inst/examples/swissmetro/plot_b23b_binary_probit.R

#!/usr/bin/env Rscript

# b23b. Binary probit model
#
# This example mirrors plot_b23b_binary_probit.py. It uses the same binary
# Train-versus-Car sample as b23a, but constructs the probit likelihood from
# native normal-CDF and indexed-selection expression nodes.

library(rbiogeme)

# prepare_swissmetro_example() is defined in example_utils.R. It parses the
# command line, validates the data/Python paths, configures the bridge, reads
# the data, and creates a fresh output directory. It does not define the model.
script_path <- commandArgs(trailingOnly = FALSE)
script_path <- sub("^--file=", "", script_path[startsWith(script_path, "--file=")][[1L]])
source(file.path(dirname(normalizePath(script_path)), "example_utils.R"))

build_b23b_binary_probit_model <- function(database) {
  # Match swissmetro_binary.py exactly: after the common Swissmetro filter,
  # remove Swissmetro (CHOICE == 2) and rows where Train or Car is unavailable.
  binary_exclude <- variable("CHOICE") == 2 |
    variable("CAR_AV_SP") == 0 |
    variable("TRAIN_AV_SP") == 0
  database <- biogeme_database_remove(database, binary_exclude)

  # Parameter names, starting values, and fixed-status flags match the native
  # Python example. None of these parameters is fixed.
  asc_car <- biogeme_beta("asc_car", start = 0)
  b_time_car <- biogeme_beta("b_time_car", start = 0)
  b_time_train <- biogeme_beta("b_time_train", start = 0)
  b_cost_car <- biogeme_beta("b_cost_car", start = 0)
  b_cost_train <- biogeme_beta("b_cost_train", start = 0)

  # The two utilities use the native alternative codes: Train = 1 and Car = 3.
  v_train <- b_time_train * variable("TRAIN_TT_SCALED") +
    b_cost_train * variable("TRAIN_COST_SCALED")
  v_car <- asc_car + b_time_car * variable("CAR_TT_SCALED") +
    b_cost_car * variable("CAR_CO_SCALED")

  # NormalCdf and Elem are symbolic nodes. The bridge compiles the complete
  # graph to native Biogeme; no R expression is evaluated row by row.
  log_probability_by_choice <- list(
    `1` = log(normal_cdf(v_train - v_car)),
    `3` = log(normal_cdf(v_car - v_train))
  )
  log_probability <- Elem(log_probability_by_choice, variable("CHOICE"))

  # Native Biogeme performs the probit likelihood, derivatives, optimization,
  # and reporting. Output files are disabled so stale results cannot be reused.
  biogeme_model(
    database = database,
    formula = log_probability,
    control = biogeme_control(
    output_directory = prepared$output,
      model_name = "b23b_probit",
      generate_html = FALSE,
      generate_yaml = FALSE,
      save_iterations = FALSE
    )
  )
}

prepared <- prepare_swissmetro_example(
  commandArgs(trailingOnly = TRUE),
  default_model = "b23b_probit"
)

database <- swissmetro_data(prepared$data)
model <- build_b23b_binary_probit_model(database)

# Estimate afresh. The native Python example may load
# saved_results/b23b_probit.yaml, but this R example never implicitly recycles
# an old YAML or iteration file.
fit <- estimate(
  model,
  model_name = "b23b_probit",
  control = model$control
)

print(summary(fit))
print(coef(fit))

invisible(fit)

Try the rbiogeme package in your browser

Any scripts or data that you put into this service are public.

rbiogeme documentation built on Sept. 29, 2026, 5:09 p.m.