inst/examples/swissmetro/plot_b07_discrete_mixture.R

#!/usr/bin/env Rscript

# b07. Discrete mixture of logit (latent-class model)
#
# This example mirrors plot_b07_discrete_mixture.py. It estimates two logit
# classes that share the ASC and cost parameters. Class 1 has no time effect;
# class 2 estimates a time coefficient. The class-1 probability is bounded
# between zero and one, and class 2 is its complement.

library(rbiogeme)

# prepare_swissmetro_example() is defined in example_utils.R. It parses the
# command line, validates the data/Python paths, configures the bridge, reads
# the data, and creates a fresh output directory. The --data, --python, and
# --output options work from any current working directory.
script_path <- commandArgs(trailingOnly = FALSE)
script_path <- sub("^--file=", "", script_path[startsWith(script_path, "--file=")][[1L]])
source(file.path(dirname(normalizePath(script_path)), "example_utils.R"))

build_b07_discrete_mixture_model <- function(database) {
  # These names, starting values, and the fixed Swissmetro ASC match Python.
  # fixed = TRUE is the R spelling of Biogeme's status flag 1.
  asc_car <- biogeme_beta("asc_car", start = 0)
  asc_train <- biogeme_beta("asc_train", start = 0)
  asc_sm <- biogeme_beta("asc_sm", start = 0, fixed = TRUE)
  b_time <- biogeme_beta("b_time", start = 0)
  b_cost <- biogeme_beta("b_cost", start = 0)

  # A bounded Beta represents the class membership probability. Writing the
  # second probability as 1 - prob_class1 makes normalization explicit while
  # leaving the complete expression tree for native Biogeme to compile.
  prob_class1 <- biogeme_beta(
    "prob_class1",
    start = 0.5,
    lower = 0,
    upper = 1
  )
  prob_class2 <- 1 - prob_class1

  train_cost <- variable("TRAIN_COST_SCALED")
  sm_cost <- variable("SM_COST_SCALED")
  car_cost <- variable("CAR_CO_SCALED")
  train_time <- variable("TRAIN_TT_SCALED")
  sm_time <- variable("SM_TT_SCALED")
  car_time <- variable("CAR_TT_SCALED")

  # Class 1 has a zero time coefficient, as in the native example.
  utilities_class_1 <- list(
    `1` = asc_train + b_cost * train_cost,
    `2` = asc_sm + b_cost * sm_cost,
    `3` = asc_car + b_cost * car_cost
  )

  # Class 2 uses the common estimated time coefficient.
  utilities_class_2 <- list(
    `1` = asc_train + b_time * train_time + b_cost * train_cost,
    `2` = asc_sm + b_time * sm_time + b_cost * sm_cost,
    `3` = asc_car + b_time * car_time + b_cost * car_cost
  )

  availability <- list(
    `1` = variable("TRAIN_AV_SP"),
    `2` = variable("SM_AV"),
    `3` = variable("CAR_AV_SP")
  )
  choice <- variable("CHOICE")

  # logit_probability() creates native models.logit expressions for each
  # class. The weighted sum is then compiled once as the log likelihood.
  probability_class_1 <- logit_probability(
    utilities = utilities_class_1,
    availability = availability,
    alternative = choice
  )
  probability_class_2 <- logit_probability(
    utilities = utilities_class_2,
    availability = availability,
    alternative = choice
  )
  log_probability <- log(
    prob_class1 * probability_class_1 + prob_class2 * probability_class_2
  )

  biogeme_model(
    database = database,
    formula = log_probability,
    control = biogeme_control(
    output_directory = prepared$output,
      model_name = "b07_discrete_mixture",
      generate_html = TRUE,
      generate_yaml = FALSE,
      save_iterations = FALSE
    )
  )
}

prepared <- prepare_swissmetro_example(
  commandArgs(trailingOnly = TRUE),
  default_model = "b07_discrete_mixture"
)

# Always estimate from the expression tree. Remove only exact b07 artifacts so
# an old YAML or iteration file cannot silently be recycled.
stale_files <- c(
  "b07_discrete_mixture.yaml",
  "__b07_discrete_mixture.iter",
  "b07_discrete_mixture.html"
)
stale_files <- file.path(prepared$output, stale_files)
stale_files <- stale_files[file.exists(stale_files)]
if (length(stale_files) > 0L) unlink(stale_files, force = TRUE)

database <- swissmetro_data(prepared$data)
model <- build_b07_discrete_mixture_model(database)

# estimate() delegates likelihood evaluation, derivatives, optimization, and
# reporting to native Biogeme; no R callback runs inside those evaluations.
fit <- estimate(
  model,
  model_name = "b07_discrete_mixture",
  control = model$control
)

print(summary(fit))
print(coef(fit))
invisible(fit)

Try the rbiogeme package in your browser

Any scripts or data that you put into this service are public.

rbiogeme documentation built on Sept. 29, 2026, 5:09 p.m.