inst/examples/swissmetro/plot_b21a_multiple_models.R

#!/usr/bin/env Rscript

# b21a. Assisted specification search
#
# This example mirrors plot_b21a_multiple_models.py. The catalog definition
# from native b21b is included here so this R script is self-contained. R
# builds only the symbolic model description; native Biogeme performs the
# assisted search, quick estimation, Pareto checkpointing, and final
# re-estimation.

library(rbiogeme)

# prepare_swissmetro_example() is defined in example_utils.R. It parses the
# command line, validates the data/Python paths, configures the bridge, reads
# the data, and creates the output directory. It does not define the model.
script_path <- commandArgs(trailingOnly = FALSE)
script_path <- sub("^--file=", "", script_path[startsWith(script_path, "--file=")][[1L]])
source(file.path(dirname(normalizePath(script_path)), "example_utils.R"))

build_b21a_multiple_models_model <- function(database) {
  # These parameters match the imported native b21b specification.
  asc_car <- biogeme_beta("asc_car", start = 0)
  asc_train <- biogeme_beta("asc_train", start = 0)
  b_time <- biogeme_beta("b_time", start = 0)
  b_cost <- biogeme_beta("b_cost", start = 0)

  # Possible segmentations for the two alternative-specific constants.
  gender_segmentation <- biogeme_database_segmentation(
    database,
    "MALE",
    c(`0` = "female", `1` = "male")
  )
  ga_segmentation <- biogeme_database_segmentation(
    database,
    "GA",
    c(`1` = "GA", `0` = "noGA"),
    reference = "noGA"
  )
  asc_segmentations <- list(gender_segmentation, ga_segmentation)
  asc_catalogs <- segmentation_catalogs(
    "asc",
    list(asc_car, asc_train),
    asc_segmentations,
    maximum_number = 2
  )
  asc_car_catalog <- asc_catalogs[[1L]]
  asc_train_catalog <- asc_catalogs[[2L]]

  # The cost catalog permits one of the GA and income segmentations.
  income_segmentation <- biogeme_database_segmentation(
    database,
    "INCOME",
    c(
      `0` = "inc-zero",
      `1` = "inc-under50",
      `2` = "inc-50-100",
      `3` = "inc-100+",
      `4` = "inc-unknown"
    )
  )
  b_cost_catalog <- segmentation_catalogs(
    "b_cost",
    list(b_cost),
    list(ga_segmentation, income_segmentation),
    maximum_number = 1
  )[[1L]]

  # Box-Cox is a native model function. It is kept as a catalog option with
  # the same lambda_time parameter used by the Python example.
  lambda_time <- biogeme_beta("lambda_time", start = 1, lower = -10, upper = 10)
  time_controller <- catalog_controller("train_tt", c("linear", "log", "boxcox"))
  train_tt_catalog <- catalog(
    "train_tt",
    list(
      linear = variable("TRAIN_TT_SCALED"),
      log = logzero(variable("TRAIN_TT_SCALED")),
      boxcox = boxcox(variable("TRAIN_TT_SCALED"), lambda_time)
    ),
    time_controller
  )
  sm_tt_catalog <- catalog(
    "sm_tt",
    list(
      linear = variable("SM_TT_SCALED"),
      log = logzero(variable("SM_TT_SCALED")),
      boxcox = boxcox(variable("SM_TT_SCALED"), lambda_time)
    ),
    time_controller
  )
  car_tt_catalog <- catalog(
    "car_tt",
    list(
      linear = variable("CAR_TT_SCALED"),
      log = logzero(variable("CAR_TT_SCALED")),
      boxcox = boxcox(variable("CAR_TT_SCALED"), lambda_time)
    ),
    time_controller
  )

  utilities <- list(
    `1` = asc_train_catalog + b_time * train_tt_catalog +
      b_cost_catalog * variable("TRAIN_COST_SCALED"),
    `2` = b_time * sm_tt_catalog + b_cost_catalog * variable("SM_COST_SCALED"),
    `3` = asc_car_catalog + b_time * car_tt_catalog +
      b_cost_catalog * variable("CAR_CO_SCALED")
  )
  availability <- list(
    `1` = variable("TRAIN_AV_SP"),
    `2` = variable("SM_AV"),
    `3` = variable("CAR_AV_SP")
  )
  log_probability <- logit_log_probability(
    utilities = utilities,
    availability = availability,
    alternative = variable("CHOICE")
  )

  biogeme_model(
    database = database,
    formula = log_probability,
    control = biogeme_control(
    output_directory = prepared$output,
      model_name = "b21_multiple_models",
      generate_html = FALSE,
      generate_yaml = FALSE,
      save_iterations = FALSE
    )
  )
}

prepared <- prepare_swissmetro_example(
  commandArgs(trailingOnly = TRUE),
  default_model = "b21_multiple_models"
)
database <- swissmetro_data(prepared$data)
model <- build_b21a_multiple_models_model(database)

# force = TRUE removes only the named native Pareto checkpoint. The bridge
# then starts the native assisted specification process from a clean state.
pareto_file <- file.path(prepared$output, "b21_multiple_models.pareto")
fit <- assisted_specification(
  model,
  objectives = "loglikelihood_dimension",
  pareto_file_name = pareto_file,
  model_name = "b21_multiple_models",
  control = model$control,
  force = TRUE
)

print(fit$summary)
for (name in names(fit$description)) {
  if (!identical(name, unname(fit$description[[name]]))) {
    aic <- fit$summary["Akaike Information Criterion", name]
    cat(sprintf("%s: %s AIC=%s\n", name, fit$description[[name]], aic))
  }
}

invisible(fit)

Try the rbiogeme package in your browser

Any scripts or data that you put into this service are public.

rbiogeme documentation built on Sept. 29, 2026, 5:09 p.m.