knitr::opts_chunk$set( collapse = TRUE, comment = "#>", # survival is suggested, only evaluate if it is installed eval = requireNamespace("survival", quietly = TRUE) )
This R Markdown document illustrates example usage of the RESIDE package using the IST dataset.
Load the RESIDE package and set a seed for reproducibility and store the folder directory for export / import.
# Load the Library library(RESIDE) # Set the seed set.seed(1234) # Store the folder path used for import / export folder_path <- tempdir()
Select the variables of interest and summarise. Selection of variables is optional, if the variables are known. They may not be known until the marginal distributions have been received.
# Select variables of interest from the IST dataset. IST_original <- IST |> dplyr::select( AGE, # AGE at Randomisation SEX, # SEX M/F RATRIAL, # Atrial Fibrillation Y/N at Randomisation # (not coded for 984 patients in the pilot phase) RSBP, # Systolic Blood Pressure at Randomisation STRK14 # Indicator of Any Stroke at 14 days ) # Convert the character variables to factors (to allow for summary) IST_original <- IST_original |> dplyr::mutate_if(is.character, factor) # Produce a summary of the variables summary(IST_original)
# Load survival and dplyr libraries library(survival) # For Cox PH model library(dplyr) # For data manipulation # Stroke event is measured at 14 days, so set this for patients IST_original$DAY <- 14 # Illustrate the 984 missing values sum(IST_original$RATRIAL == "") # Remove the missing values IST_original <- IST_original[!IST_original$RATRIAL == "",] # Drop the factor name for the missing values IST_original$RATRIAL <- droplevels(IST_original$RATRIAL) # Summarise the variable to show there are no longer missing values summary(IST_original$RATRIAL) # Fit a Cox PH model cox.ph <- coxph(Surv(DAY, STRK14) ~ AGE + SEX + RATRIAL + RSBP, data = IST_original) # Output the summary of the Cox PH Model cox.ph
Use the get_marginal_distributions() function to get the marginal distributions, additionally selecting which variables using the variables parameter.
# Get the Marginal Distributions for the selected variables marginals <- get_marginal_distributions( IST, variables = c( "AGE", "SEX", "RATRIAL", "RSBP", "STRK14" ) )
Export the marginal distributions using the export_marginal_distributions() function, using the force parameter to override any existing files.
# Export the Marginal Distributions export_marginal_distributions(marginals, folder_path = folder_path, force = TRUE)
Import the exported marginal distributions using the import_marginal_distributions() function
# Import the Marginal Distributions imported_marginals <- import_marginal_distributions(folder_path = folder_path)
Synthesise data from the imported marginals using the synthesise_data function.
# Synthesise a dataset from the imported Marginal Distributions (without correlations) sim_df <- synthesise_data(imported_marginals)
Summarise the simulated data
# Convert any Character variables to Factors sim_df <- sim_df |> dplyr::mutate_if(is.character, factor) # Summarise the synthesised data summary(sim_df)
Fit the same cox model as earlier except this time on the simulated data.
# As before the events are measured at day 14 sim_df$DAY <- 14 # Show that the missing observations are in the data sum(sim_df$RATRIAL == "") # Remove the missing observations sim_df <- sim_df[!sim_df$RATRIAL == "",] # Remove the missing factor name sim_df$RATRIAL <- droplevels(sim_df$RATRIAL) # Show that there are no missing observations summary(sim_df$RATRIAL) # Fit the model on the synthesised data cox.ph.sim <- coxph(Surv(DAY, STRK14) ~ AGE + SEX + RATRIAL + RSBP, data = sim_df) # Show a summary of the model cox.ph.sim
Synthesise data from the imported marginals with correlations, using the correlations parameter to specify a list of assumed correlations created with the correlation() function. Categorical variables are correlated using a single category, specified with factor_name.x or factor_name.y.
# Synthesise data specifying assumed correlations sim_df_cor <- synthesise_data( imported_marginals, correlations = list( # Patients without atrial fibrillation are less likely to have a stroke correlation("RATRIAL", "STRK14", -0.2, factor_name.x = "N"), # Older patients have a higher systolic blood pressure correlation("AGE", "RSBP", 0.3) ) )
Summarise the synthesised data (with correlations)
# Convert to Factors from Character variables sim_df_cor <- sim_df_cor |> dplyr::mutate_if(is.character, factor) # Summarise the synthesised dataset summary(sim_df_cor)
Using the synthesised data (with correlations) fit a Cox PH model, with the same parameters as earlier.
# Again events are measured at 14 days sim_df_cor$DAY <- 14 # Again check that the missing values where added sum(sim_df_cor$RATRIAL == "") # Again remove the missing values sim_df_cor <- sim_df_cor[!sim_df_cor$RATRIAL == "",] # Again drop the missing factor sim_df_cor$RATRIAL <- droplevels(sim_df_cor$RATRIAL) # Show there are no missing values summary(sim_df_cor$RATRIAL) # Fit the model on the synthesised data (with correlations) cox.ph.sim.cor <- coxph(Surv(DAY, STRK14) ~ AGE + SEX + RATRIAL + RSBP, data = sim_df_cor) # Show a summary of the model cox.ph.sim.cor
Any scripts or data that you put into this service are public.
Add the following code to your website.
For more information on customizing the embed code, read Embedding Snippets.