inst/examples/swissmetro/plot_b04_validation.R

#!/usr/bin/env Rscript

# b04. Out-of-sample validation
#
# This example estimates the Swissmetro MNL and then performs five-fold
# out-of-sample validation. Native Biogeme re-estimates the model on each
# training fold and evaluates the resulting model on the held-out fold.

library(rbiogeme)

# The shared helper contains command-line parsing and data preparation. The
# complete model specification remains in this script.
script_path <- commandArgs(trailingOnly = FALSE)
script_path <- sub("^--file=", "", script_path[startsWith(script_path, "--file=")][[1L]])
source(file.path(dirname(normalizePath(script_path)), "example_utils.R"))

build_b04_validation_model <- function(database) {
  # Each biogeme_beta() call creates a symbolic native parameter. The
  # Swissmetro ASC is fixed at zero to identify the model.
  asc_car <- biogeme_beta("asc_car", start = 0)
  asc_train <- biogeme_beta("asc_train", start = 0)
  asc_sm <- biogeme_beta("asc_sm", start = 0, fixed = TRUE)
  b_time <- biogeme_beta("b_time", start = 0)
  b_cost <- biogeme_beta("b_cost", start = 0)

  # variable() and the overloaded arithmetic operators build a symbolic
  # expression tree. They do not evaluate rows in R.
  logit_model(
    database = database,
    choice = "CHOICE",
    utilities = list(
      `1` = asc_train + b_time * variable("TRAIN_TT_SCALED") +
        b_cost * variable("TRAIN_COST_SCALED"),
      `2` = asc_sm + b_time * variable("SM_TT_SCALED") +
        b_cost * variable("SM_COST_SCALED"),
      `3` = asc_car + b_time * variable("CAR_TT_SCALED") +
        b_cost * variable("CAR_CO_SCALED")
    ),
    availability = list(
      `1` = variable("TRAIN_AV_SP"),
      `2` = variable("SM_AV"),
      `3` = variable("CAR_AV_SP")
    )
  )
}

# prepare_swissmetro_example() is defined in example_utils.R. It parses the
# command line, validates the data/Python paths, configures the bridge, reads
# the data, and creates a fresh output directory. The --data, --python,
# --output, and --seed options work from any current working directory.
prepared <- prepare_swissmetro_example(
  commandArgs(trailingOnly = TRUE),
  default_model = "b04_validation"
)

# estimate() always performs fresh native estimation. Remove only exact b04
# artifacts so a reused output directory cannot silently recycle old results.
stale_files <- c(
  "b04_validation.yaml",
  "__b04_validation.iter",
  "b04_validation.html",
  "rbiogeme_validation.html"
)
stale_files <- file.path(prepared$output, stale_files)
stale_files <- stale_files[file.exists(stale_files)]
if (length(stale_files) > 0L) unlink(stale_files, force = TRUE)

database <- swissmetro_data(prepared$data)
model <- build_b04_validation_model(database)
control <- biogeme_control(
    output_directory = prepared$output,
  model_name = "b04_validation",
  generate_html = TRUE,
  generate_yaml = FALSE,
  save_iterations = FALSE
)

# Estimate the full-data model first, as in the native example. The public
# validate() wrapper is defined by rbiogeme and delegates fold construction,
# re-estimation, and held-out evaluation to native BIOGEME.validate().
fit <- estimate(model, model_name = "b04_validation", control = control)

# The native example does not choose a seed, so its folds vary between runs.
# This R example uses an explicit default seed for reproducible fold assignment;
# pass --seed=<integer> to select another native NumPy fold seed.
seed <- if (!is.null(prepared$options$seed) && nzchar(prepared$options$seed)) {
  example_integer(prepared$options$seed, "seed")
} else {
  73129L
}
validation_results <- validate(
  model = model,
  fit = fit,
  folds = 5L,
  seed = seed,
  control = control
)

# Each fold contains the native simulated log-likelihood contributions for its
# held-out observations. Sum the first (and only) native formula, matching the
# reporting loop in plot_b04_validation.py.
for (fold in validation_results) {
  values <- fold$simulated_values
  values <- if (is.data.frame(values)) values else as.data.frame(values, check.names = FALSE)
  log_likelihood <- sum(values[[1L]])
  cat(
    sprintf(
      "Log likelihood for %d validation data: %.15g\n",
      nrow(values),
      log_likelihood
    )
  )
}

print(summary(fit))
invisible(validation_results)

Try the rbiogeme package in your browser

Any scripts or data that you put into this service are public.

rbiogeme documentation built on Sept. 29, 2026, 5:09 p.m.