inst/examples/sampling/plot_b01logit.R

#!/usr/bin/env Rscript

# Logit estimation using native sampling of alternatives.
#
# This script is self-contained: it defines the utility, the cross-variable,
# the partition, and the estimation call. The bridge delegates alternative
# sampling and the sampled likelihood to native Biogeme.

library(rbiogeme)

# prepare_sampling_example() is defined in this directory's example_utils.R.
# It reads the two explicit input files and creates the output directory; it
# does not define the model.
script_path <- commandArgs(trailingOnly = FALSE)
script_path <- sub("^--file=", "", script_path[startsWith(script_path, "--file=")][[1L]])
source(file.path(dirname(normalizePath(script_path)), "example_utils.R"))

prepared <- prepare_sampling_example(
  commandArgs(trailingOnly = TRUE),
  default_model = "logit_asian_10_alt"
)
alternatives <- prepared$alternatives
observations <- prepared$observations

# The native example samples 10 alternatives from the Asian/non-Asian
# partition of the 100 restaurants, using logit_4 as the observed choice.
all_alternatives <- sort(unique(as.integer(alternatives$ID)))
asian <- all_alternatives[alternatives$Asian[match(all_alternatives, alternatives$ID)] == 1]
partition <- biogeme_sampling_partition(
  segments = list(asian, setdiff(all_alternatives, asian)),
  sample_sizes = sampling_segment_sizes(10, 2),
  full_set = all_alternatives
)

# CrossVariableTuple is represented by cross_variable(). Native Biogeme
# expands this individual/alternative formula after each sample is drawn.
log_dist <- cross_variable(
  "log_dist",
  log(((variable("user_lat") - variable("rest_lat"))^2 +
    (variable("user_lon") - variable("rest_lon"))^2)^0.5)
)

# Complete utility specification. Names are preserved exactly from Python.
beta_rating <- biogeme_beta("beta_rating", start = 0)
beta_price <- biogeme_beta("beta_price", start = 0)
beta_chinese <- biogeme_beta("beta_chinese", start = 0)
beta_japanese <- biogeme_beta("beta_japanese", start = 0)
beta_korean <- biogeme_beta("beta_korean", start = 0)
beta_indian <- biogeme_beta("beta_indian", start = 0)
beta_french <- biogeme_beta("beta_french", start = 0)
beta_mexican <- biogeme_beta("beta_mexican", start = 0)
beta_lebanese <- biogeme_beta("beta_lebanese", start = 0)
beta_ethiopian <- biogeme_beta("beta_ethiopian", start = 0)
beta_log_dist <- biogeme_beta("beta_log_dist", start = 0)
utility <- beta_rating * variable("rating") +
  beta_price * variable("price") +
  beta_chinese * variable("category_Chinese") +
  beta_japanese * variable("category_Japanese") +
  beta_korean * variable("category_Korean") +
  beta_indian * variable("category_Indian") +
  beta_french * variable("category_French") +
  beta_mexican * variable("category_Mexican") +
  beta_lebanese * variable("category_Lebanese") +
  beta_ethiopian * variable("category_Ethiopian") +
  beta_log_dist * variable("log_dist")

control <- biogeme_control(
    output_directory = prepared$output,
  generate_html = FALSE,
  generate_yaml = FALSE,
  save_iterations = FALSE
)
model <- sampled_alternatives_model(
  alternatives = alternatives,
  individuals = observations,
  choice_column = "logit_4",
  id_column = "ID",
  utility = utility,
  combined_variables = list(log_dist),
  partition = partition,
  biogeme_file_name = file.path(prepared$output, "logit_asian_10_alt.dat"),
  model_type = "logit",
  control = control
)

fit <- estimate_sampled_alternatives(
  model,
  model_name = "logit_asian_10_alt",
  control = control
)
print(fit)

# Native compare.py compares estimates with the synthetic data-generating
# values. The table is post-processing only; estimation remains native.
comparison <- compare_sampling_parameters(fit, sampling_true_parameters())
print(comparison$data)
print(comparison$message)

invisible(fit)

Try the rbiogeme package in your browser

Any scripts or data that you put into this service are public.

rbiogeme documentation built on Sept. 29, 2026, 5:09 p.m.