Nothing
#!/usr/bin/env Rscript
# b13. Simulation of a panel model
#
# This example mirrors plot_b13_panel_simul.py. It first estimates the b12
# panel model in a clean run, then evaluates the complete panel log likelihood
# as one native aggregated scalar and simulates native Numerator and
# Denominator expressions for individual-level time parameters.
library(rbiogeme)
# prepare_swissmetro_example() is defined in example_utils.R. It parses the
# command line, validates the data/Python paths, configures the bridge, reads
# the data, and creates a fresh output directory. The --data, --python,
# --output, --draws, --estimation-draws, and --seed options work from any
# current working directory.
script_path <- commandArgs(trailingOnly = FALSE)
script_path <- sub("^--file=", "", script_path[startsWith(script_path, "--file=")][[1L]])
source(file.path(dirname(normalizePath(script_path)), "example_utils.R"))
build_panel_simulation_components <- function(
database,
number_of_draws,
seed,
b12_bounds = FALSE,
with_simulations = FALSE,
model_name = "b13_panel_simul"
) {
# b13 uses unbounded parameters for simulation. The b12 estimation model
# uses the exact bounds from plot_b12_panel.py, whose result is the native
# starting point for this example.
if (b12_bounds) {
b_cost <- biogeme_beta("b_cost", start = 0, upper = 0)
b_time <- biogeme_beta("b_time", start = 0, upper = 0)
b_time_s <- biogeme_beta("b_time_s", start = 1, lower = 1.0e-5)
asc_car_s <- biogeme_beta("asc_car_s", start = 1, lower = 1.0e-5)
asc_train_s <- biogeme_beta("asc_train_s", start = 1, lower = 1.0e-5)
asc_sm_s <- biogeme_beta("asc_sm_s", start = 1, lower = 1.0e-5)
} else {
b_cost <- biogeme_beta("b_cost", start = 0)
b_time <- biogeme_beta("b_time", start = 0)
b_time_s <- biogeme_beta("b_time_s", start = 1)
asc_car_s <- biogeme_beta("asc_car_s", start = 1)
asc_train_s <- biogeme_beta("asc_train_s", start = 1)
asc_sm_s <- biogeme_beta("asc_sm_s", start = 1)
}
asc_car <- biogeme_beta("asc_car", start = 0)
asc_train <- biogeme_beta("asc_train", start = 0)
asc_sm <- biogeme_beta("asc_sm", start = 0)
# Named Draws nodes are generated and reused by native Biogeme. In the
# panel database, the same individual draw is used throughout a trajectory.
b_time_rnd <- b_time + b_time_s * draw("b_time_rnd", "NORMAL_ANTI")
asc_car_rnd <- asc_car + asc_car_s * draw("asc_car_rnd", "NORMAL_ANTI")
asc_train_rnd <- asc_train + asc_train_s * draw("asc_train_rnd", "NORMAL_ANTI")
asc_sm_rnd <- asc_sm + asc_sm_s * draw("asc_sm_rnd", "NORMAL_ANTI")
utilities <- list(
`1` = asc_train_rnd + b_time_rnd * variable("TRAIN_TT_SCALED") +
b_cost * variable("TRAIN_COST_SCALED"),
`2` = asc_sm_rnd + b_time_rnd * variable("SM_TT_SCALED") +
b_cost * variable("SM_COST_SCALED"),
`3` = asc_car_rnd + b_time_rnd * variable("CAR_TT_SCALED") +
b_cost * variable("CAR_CO_SCALED")
)
availability <- list(
`1` = variable("TRAIN_AV_SP"),
`2` = variable("SM_AV"),
`3` = variable("CAR_AV_SP")
)
kernel <- logit_probability(
utilities = utilities,
availability = availability,
alternative = variable("CHOICE")
)
trajectory_probability <- panel_likelihood_trajectory(kernel)
log_probability <- log(monte_carlo(trajectory_probability))
simulations <- NULL
if (with_simulations) {
# These expressions are evaluated row-by-row at the panel-trajectory
# level by native BIOGEME. The individual parameter is formed only after
# the native Numerator and Denominator columns have been returned.
simulations <- list(
Numerator = monte_carlo(b_time_rnd * trajectory_probability),
Denominator = monte_carlo(trajectory_probability)
)
}
draws <- list(
biogeme_draws("b_time_rnd", "NORMAL_ANTI", number_of_draws, seed),
biogeme_draws("asc_car_rnd", "NORMAL_ANTI", number_of_draws, seed),
biogeme_draws("asc_train_rnd", "NORMAL_ANTI", number_of_draws, seed),
biogeme_draws("asc_sm_rnd", "NORMAL_ANTI", number_of_draws, seed)
)
list(
model = biogeme_model(
database = database,
formula = log_probability,
simulations = simulations,
draws = draws,
control = biogeme_control(
output_directory = prepared$output,
model_name = model_name,
number_of_draws = number_of_draws,
seed = seed,
generate_html = FALSE,
generate_yaml = FALSE,
save_iterations = FALSE
)
),
log_probability = log_probability,
simulations = simulations,
draws = draws
)
}
prepared <- prepare_swissmetro_example(
commandArgs(trailingOnly = TRUE),
default_model = "b13_panel_simul"
)
number_of_draws <- if (!is.null(prepared$options$draws) && nzchar(prepared$options$draws)) {
example_integer(prepared$options$draws, "draws")
} else {
100000L
}
estimation_draws <- if (!is.null(prepared$options$estimation_draws) &&
nzchar(prepared$options$estimation_draws)) {
example_integer(prepared$options$estimation_draws, "estimation-draws")
} else {
5000L
}
seed <- if (!is.null(prepared$options$seed) && nzchar(prepared$options$seed)) {
example_integer(prepared$options$seed, "seed")
} else {
1223L
}
# Always estimate b12 afresh and remove only exact native artifacts. No old
# YAML or iteration file can silently supply the estimates used below.
stale_files <- c(
"b12_panel.yaml",
"__b12_panel.iter",
"b12_panel.html",
"b13_panel_simul.yaml",
"__b13_panel_simul.iter",
"b13_panel_simul.html"
)
stale_files <- file.path(prepared$output, stale_files)
stale_files <- stale_files[file.exists(stale_files)]
if (length(stale_files) > 0L) unlink(stale_files, force = TRUE)
# panel=TRUE declares ID as the panel identifier after the native-equivalent
# Swissmetro filter and derived-variable operations. The bridge validates that
# observations for each ID remain contiguous before constructing native data.
database <- swissmetro_data(prepared$data, panel = TRUE)
estimation_components <- build_panel_simulation_components(
database,
number_of_draws = estimation_draws,
seed = seed,
b12_bounds = TRUE,
with_simulations = FALSE,
model_name = "b12_panel"
)
fit <- estimate(
estimation_components$model,
model_name = "b12_panel",
control = estimation_components$model$control
)
print(summary(fit))
print(coef(fit))
simulation_components <- build_panel_simulation_components(
database,
number_of_draws = number_of_draws,
seed = seed,
b12_bounds = FALSE,
with_simulations = TRUE,
model_name = "b13_panel_simul"
)
# This calls Biogeme's public single-formula JAX evaluator. It returns the
# scalar aggregate directly; R does not sum or otherwise evaluate the panel
# likelihood locally.
simulated_loglike <- simulate_single_formula(
simulation_components$model,
expression = simulation_components$log_probability,
beta = fit,
number_of_draws = number_of_draws,
seed = seed,
numerically_safe = FALSE,
use_jit = TRUE
)
cat(sprintf("Simulated log likelihood: %.12f\n", simulated_loglike))
# simulate() compiles the named expression dictionary once and delegates all
# panel aggregation, Monte Carlo integration, and draw handling to Biogeme.
simulation <- simulate(
simulation_components$model,
expressions = simulation_components$simulations,
beta = fit,
control = biogeme_control(
output_directory = prepared$output,
model_name = "b13_panel_simul",
number_of_draws = number_of_draws,
seed = seed,
generate_html = FALSE,
generate_yaml = FALSE,
save_iterations = FALSE
)
)
simulation_values <- as.data.frame(simulation, check.names = FALSE)
simulation_values[["Individual-level parameters"]] <-
simulation_values[["Numerator"]] / simulation_values[["Denominator"]]
cat(sprintf(
"Simulation dimensions: %d rows x %d columns\n",
nrow(simulation_values),
ncol(simulation_values)
))
print(utils::head(simulation_values))
report_database <- biogeme_database_materialize(database)
cat(sprintf("Panel identifier: %s\n", database$panel_id))
cat(sprintf("Filtered observations: %d\n", nrow(report_database$data)))
cat(sprintf("Draws per individual: %d\n", number_of_draws))
invisible(list(fit = fit, simulated_loglike = simulated_loglike, simulation = simulation_values))
Any scripts or data that you put into this service are public.
Add the following code to your website.
For more information on customizing the embed code, read Embedding Snippets.