## ----include = FALSE----------------------------------------------------------
knitr::opts_chunk$set(
  collapse = TRUE,
  comment = "#>",
  # survival is suggested, only evaluate if it is installed
  eval = requireNamespace("survival", quietly = TRUE)
)

## ----load-library-------------------------------------------------------------
# Load the Library
library(RESIDE)
# Set the seed
set.seed(1234)
# Store the folder path used for import / export
folder_path <- tempdir()

## ----summarise-data-----------------------------------------------------------
# Select variables of interest from the IST dataset.
IST_original <- IST |> dplyr::select(
  AGE, # AGE at Randomisation
  SEX, # SEX M/F
  RATRIAL, # Atrial Fibrillation Y/N at Randomisation 
  # (not coded for 984 patients in the pilot phase)
  RSBP, # Systolic Blood Pressure at Randomisation
  STRK14 # Indicator of Any Stroke at 14 days
)

# Convert the character variables to factors (to allow for summary)
IST_original <- IST_original |> dplyr::mutate_if(is.character, factor)

# Produce a summary of the variables
summary(IST_original)


## ----original-model-----------------------------------------------------------
# Load survival and dplyr libraries
library(survival) # For Cox PH model
library(dplyr) # For data manipulation

# Stroke event is measured at 14 days, so set this for patients
IST_original$DAY <- 14

# Illustrate the 984 missing values
sum(IST_original$RATRIAL == "")

# Remove the missing values
IST_original <- IST_original[!IST_original$RATRIAL == "",]

# Drop the factor name for the missing values
IST_original$RATRIAL <- droplevels(IST_original$RATRIAL)

# Summarise the variable to show there are no longer missing values
summary(IST_original$RATRIAL)

# Fit a Cox PH model
cox.ph <- coxph(Surv(DAY, STRK14) ~ AGE + SEX + RATRIAL + RSBP, data = IST_original) 

# Output the summary of the Cox PH Model
cox.ph

## ----get-marginals------------------------------------------------------------
# Get the Marginal Distributions for the selected variables
marginals <- get_marginal_distributions(
  IST,
  variables = c(
    "AGE",
    "SEX",
    "RATRIAL",
    "RSBP",
    "STRK14"
  )
)


## ----export-marginals---------------------------------------------------------
# Export the Marginal Distributions
export_marginal_distributions(marginals,
                              folder_path = folder_path,
                              force = TRUE)

## ----import-marginals---------------------------------------------------------
# Import the Marginal Distributions
imported_marginals <- import_marginal_distributions(folder_path = folder_path)

## ----synthesis-data-----------------------------------------------------------
# Synthesise a dataset from the imported Marginal Distributions (without correlations)
sim_df <- synthesise_data(imported_marginals)

## ----summarise-sim-data-------------------------------------------------------
# Convert any Character variables to Factors
sim_df <- sim_df |> dplyr::mutate_if(is.character, factor)
# Summarise the synthesised data
summary(sim_df)

## ----sim-data-model-----------------------------------------------------------

# As before the events are measured at day 14
sim_df$DAY <- 14

# Show that the missing observations are in the data
sum(sim_df$RATRIAL == "")

# Remove the missing observations
sim_df <- sim_df[!sim_df$RATRIAL == "",]

# Remove the missing factor name
sim_df$RATRIAL <- droplevels(sim_df$RATRIAL)

# Show that there are no missing observations
summary(sim_df$RATRIAL)

# Fit the model on the synthesised data
cox.ph.sim <- coxph(Surv(DAY, STRK14) ~ AGE + SEX + RATRIAL + RSBP, data = sim_df) 

# Show a summary of the model
cox.ph.sim

## ----synthesis-data-cor-------------------------------------------------------
# Synthesise data specifying assumed correlations
sim_df_cor <- synthesise_data(
  imported_marginals,
  correlations = list(
    # Patients without atrial fibrillation are less likely to have a stroke
    correlation("RATRIAL", "STRK14", -0.2, factor_name.x = "N"),
    # Older patients have a higher systolic blood pressure
    correlation("AGE", "RSBP", 0.3)
  )
)

## ----summarise-sim-data-cor---------------------------------------------------
# Convert to Factors from Character variables
sim_df_cor <- sim_df_cor |> dplyr::mutate_if(is.character, factor)
# Summarise the synthesised dataset
summary(sim_df_cor)

## ----sim-data-model-cor-------------------------------------------------------

# Again events are measured at 14 days
sim_df_cor$DAY <- 14

# Again check that the missing values where added
sum(sim_df_cor$RATRIAL == "")

# Again remove the missing values
sim_df_cor <- sim_df_cor[!sim_df_cor$RATRIAL == "",]

# Again drop the missing factor
sim_df_cor$RATRIAL <- droplevels(sim_df_cor$RATRIAL)

# Show there are no missing values
summary(sim_df_cor$RATRIAL)

# Fit the model on the synthesised data (with correlations)
cox.ph.sim.cor <- coxph(Surv(DAY, STRK14) ~ AGE + SEX + RATRIAL + RSBP, data = sim_df_cor)

# Show a summary of the model
cox.ph.sim.cor

