tests/testthat/test-swissmetro-b22a.R

native_b22a_specification <- function(data_path, pareto_file_name, random_seed) {
  native_examples <- "/Users/bierlair/MyFiles/github/biogeme/docs/source/examples/swissmetro"
  sys <- reticulate::import("sys", convert = FALSE)
  sys$path$insert(0L, native_examples)
  old_directory <- getwd()
  setwd(dirname(pareto_file_name))
  on.exit(setwd(old_directory), add = TRUE)
  file.copy(data_path, file.path(getwd(), "swissmetro.dat"), overwrite = TRUE)

  native_environment <- new.env(parent = emptyenv())
  reticulate::source_python(
    file.path(native_examples, "plot_b22b_multiple_models_spec.py"),
    envir = native_environment
  )
  native_biogeme <- native_environment$the_biogeme
  assisted_module <- reticulate::import("biogeme.assisted", convert = FALSE)
  objectives_module <- reticulate::import("biogeme.multiobjectives", convert = FALSE)
  catalog_module <- reticulate::import("biogeme.catalog", convert = FALSE)
  specification <- reticulate::import(
    "biogeme.catalog.specification",
    convert = FALSE
  )$Specification
  specification$all_results <- reticulate::dict()
  specification$model_names <- NULL
  reticulate::import("random", convert = FALSE)$seed(as.integer(random_seed))

  assisted <- assisted_module$AssistedSpecification(
    biogeme_object = native_biogeme,
    multi_objectives = objectives_module$aic_bic_dimension,
    pareto_file_name = pareto_file_name
  )
  results <- assisted$run()
  bridge <- rbiogeme:::biogeme_bridge()
  keys <- vapply(
    reticulate::iterate(results$keys()),
    as.character,
    character(1)
  )
  serialized <- lapply(keys, function(key) {
    reticulate::py_to_r(
      bridge$extract_estimation_results(reticulate::py_get_item(results, key))
    )
  })
  names(serialized) <- keys
  list(
    count = as.integer(reticulate::py_to_r(
      catalog_module$count_number_of_specifications(native_biogeme$log_like)
    )),
    results = serialized,
    pareto_statistics = vapply(
      reticulate::iterate(assisted$pareto$statistics()),
      as.character,
      character(1)
    )
  )
}

build_b22a_test_model <- function(database) {
  asc_car <- biogeme_beta("asc_car", start = 0)
  asc_train <- biogeme_beta("asc_train", start = 0)
  b_time <- biogeme_beta("b_time", start = 0)
  b_cost <- biogeme_beta("b_cost", start = 0)
  b_headway <- biogeme_beta("b_headway", start = 0)
  gender <- biogeme_database_segmentation(
    database,
    "MALE",
    c(`0` = "female", `1` = "male")
  )
  ga <- biogeme_database_segmentation(
    database,
    "GA",
    c(`0` = "without_ga", `1` = "with_ga")
  )
  luggage <- biogeme_database_segmentation(
    database,
    "LUGGAGE",
    c(`0` = "no_lugg", `1` = "one_lugg", `3` = "several_lugg")
  )
  asc_catalogs <- segmentation_catalogs(
    "asc",
    list(asc_car, asc_train),
    list(gender, luggage, ga),
    maximum_number = 2
  )
  headway_controller <- catalog_controller(
    "train_headway_catalog",
    c("without_headway", "with_headway")
  )
  train_headway <- catalog(
    "train_headway_catalog",
    list(
      without_headway = 0,
      with_headway = b_headway * variable("TRAIN_HE")
    ),
    headway_controller
  )
  sm_headway <- catalog(
    "sm_headway_catalog",
    list(
      without_headway = 0,
      with_headway = b_headway * variable("SM_HE")
    ),
    headway_controller
  )
  lambda_tt <- biogeme_beta("lambda_tt", start = 1, lower = -10, upper = 10)
  time_controller <- catalog_controller(
    "train_tt_catalog",
    c("linear", "log", "sqrt", "piecewise_1", "piecewise_2", "boxcox")
  )
  time_options <- function(name) {
    x <- variable(name)
    list(
      linear = x,
      log = logzero(x),
      sqrt = x ^ 0.5,
      piecewise_1 = piecewise(x, list(0, 0.1, NULL)),
      piecewise_2 = piecewise(x, list(0, 0.25, NULL)),
      boxcox = boxcox(x, lambda_tt)
    )
  }
  train_tt <- catalog("train_tt_catalog", time_options("TRAIN_TT_SCALED"), time_controller)
  sm_tt <- catalog("sm_tt_catalog", time_options("SM_TT_SCALED"), time_controller)
  car_tt <- catalog("car_tt_catalog", time_options("CAR_TT_SCALED"), time_controller)
  lambda_cost <- biogeme_beta("lambda_cost", start = 1, lower = -10, upper = 10)
  cost_controller <- catalog_controller(
    "train_cost_catalog",
    c("linear", "log", "sqrt", "piecewise_1", "piecewise_2", "boxcox")
  )
  cost_options <- function(name) {
    x <- variable(name)
    list(
      linear = x,
      log = logzero(x),
      sqrt = x ^ 0.5,
      piecewise_1 = piecewise(x, list(0, 0.1, NULL)),
      piecewise_2 = piecewise(x, list(0, 0.25, NULL)),
      boxcox = boxcox(x, lambda_cost)
    )
  }
  train_cost <- catalog("train_cost_catalog", cost_options("TRAIN_COST_SCALED"), cost_controller)
  sm_cost <- catalog("sm_cost_catalog", cost_options("SM_COST_SCALED"), cost_controller)
  car_cost <- catalog("car_cost_catalog", cost_options("CAR_CO_SCALED"), cost_controller)
  utilities <- list(
    `1` = asc_catalogs[[2L]] + b_time * train_tt + b_cost * train_cost + train_headway,
    `2` = b_time * sm_tt + b_cost * sm_cost + sm_headway,
    `3` = asc_catalogs[[1L]] + b_time * car_tt + b_cost * car_cost
  )
  biogeme_model(
    database = database,
    formula = logit_log_probability(
      utilities = utilities,
      availability = list(
        `1` = variable("TRAIN_AV_SP"),
        `2` = variable("SM_AV"),
        `3` = variable("CAR_AV_SP")
      ),
      alternative = variable("CHOICE")
    ),
    control = biogeme_control(
      model_name = "b22_multiple_models",
      generate_html = FALSE,
      generate_yaml = FALSE,
      save_iterations = FALSE
    )
  )
}

test_that("b22a Swissmetro large assisted specification matches native Biogeme", {
  skip_if_not(
    identical(Sys.getenv("RBIOGEME_RUN_INTEGRATION"), "1"),
    "Set RBIOGEME_RUN_INTEGRATION=1 to run full Swissmetro equivalence tests"
  )
  skip_if_not(
    rbiogeme_test_configure_python(),
    "Set RBIOGEME_PYTHON to a compatible native Biogeme interpreter"
  )
  data_path <- rbiogeme_test_swissmetro_path()
  skip_if(!nzchar(data_path), "Set RBIOGEME_SWISSMETRO_DATA to the Swissmetro .dat file")

  data <- read.delim(data_path, check.names = FALSE, stringsAsFactors = FALSE)
  temporary_directory <- tempfile("rbiogeme-b22a-")
  dir.create(temporary_directory, recursive = TRUE)
  original_directory <- getwd()
  setwd(temporary_directory)
  on.exit(setwd(original_directory), add = TRUE)
  random_seed <- 220826L
  native_pareto_file <- file.path(getwd(), "native_b22_multiple_models.pareto")
  native <- native_b22a_specification(data_path, native_pareto_file, random_seed)

  database <- swissmetro_data(data)
  model <- build_b22a_test_model(database)
  expect_equal(native$count, 504L)
  expect_equal(
    count_number_of_specifications(
      model,
      model_name = "b22_multiple_models",
      control = model$control
    ),
    native$count
  )

  # VNS uses Python's random module for neighborhood selection. Resetting the
  # same native seed makes this comparison reproducible while preserving the
  # heuristic algorithm itself. The heuristic is exercised independently;
  # because native VNS can take different branches after tiny optimizer
  # differences, its result set is compared below through the shared native
  # Pareto checkpoint instead of requiring identical search history.
  r_pareto_file <- file.path(getwd(), "r_b22_multiple_models.pareto")
  reticulate::import("random", convert = FALSE)$seed(as.integer(random_seed))
  r_assisted <- assisted_specification(
    model,
    objectives = "aic_bic_dimension",
    pareto_file_name = r_pareto_file,
    model_name = "b22_multiple_models",
    control = model$control,
    force = TRUE
  )

  expect_gt(length(r_assisted$results), 0L)
  expect_true(all(vapply(r_assisted$results, inherits, logical(1), what = "biogeme_fit")))

  # Re-process the native checkpoint through the R interface. This compares
  # the same Pareto configurations and isolates expression compilation from
  # the intentionally heuristic VNS path.
  r_fit <- pareto_post_processing(
    model,
    pareto_file_name = native_pareto_file,
    model_name = "b22_multiple_models",
    control = model$control,
    recycle = FALSE
  )
  expect_equal(length(r_fit$results), length(native$results))
  expect_equal(r_fit$pareto_statistics, native$pareto_statistics)
  expect_setequal(names(r_fit$results), names(native$results))
  for (configuration in names(native$results)) {
    r_result <- r_fit$results[[configuration]]
    native_result <- native$results[[configuration]]
    expect_identical(r_result$beta_names, native_result$beta_names, info = configuration)
    expect_equal(unname(coef(r_result)), native_result$beta_values, tolerance = 1e-6, info = configuration)
    expect_equal(
      as.numeric(logLik(r_result)),
      native_result$final_log_likelihood,
      tolerance = 1e-6,
      info = configuration
    )
  }
})

Try the rbiogeme package in your browser

Any scripts or data that you put into this service are public.

rbiogeme documentation built on Sept. 29, 2026, 5:09 p.m.