## ----setup, include=FALSE-----------------------------------------------------
knitr::opts_chunk$set(
  collapse  = TRUE,
  comment   = "#>",
  fig.width = 7,
  fig.height = 4,
  warning   = FALSE,
  message   = FALSE
)

## ----load---------------------------------------------------------------------
library(mudnester)

## ----synthetic-data-----------------------------------------------------------
set.seed(42)
n <- 80

# Case notifications (linelist style)
cases_raw <- data.frame(
  identity        = paste0("PT", seq_len(n)),
  first_name      = sample(c("James", "Sarah", "Michael", "Emma", "William"), n, TRUE),
  surname         = sample(c("Smith", "Jones", "Williams", "Taylor", "Brown"), n, TRUE),
  date_of_birth   = as.Date("1980-01-01") + sample(-10000:10000, n, TRUE),
  date_of_onset   = as.Date("2024-01-01") + sample(0:364, n, TRUE),
  disease_name    = sample(c("COVID-19", "Influenza A", "RSV"), n, TRUE),
  gender          = sample(c("M", "F"), n, TRUE),
  postcode        = sample(c("4556", "4557", "4558", "4560"), n, TRUE),
  medicare_no     = paste0(sample(1000:9999, n, TRUE), sample(10000:99999, n, TRUE)),
  indigenous_status = sample(c("Non-Indigenous", "Aboriginal", "Torres Strait Islander",
                                "Unknown"), n, TRUE, prob = c(0.85, 0.08, 0.02, 0.05)),
  stringsAsFactors = FALSE
)

# Hospital admissions
hosp_raw <- data.frame(
  patient_id     = paste0("UR", seq_len(50)),
  firstname      = sample(c("James", "Sarah", "Michael"), 50, TRUE),
  last_name      = sample(c("Smith", "Jones", "Williams"), 50, TRUE),
  birth_date     = as.Date("1950-01-01") + sample(-5000:5000, 50, TRUE),
  date_of_admission = as.Date("2024-01-01") + sample(0:364, 50, TRUE),
  date_of_discharge = as.Date("2024-01-01") + sample(1:400, 50, TRUE),
  medicare_number = paste0(sample(1000:9999, 50, TRUE), sample(10000:99999, 50, TRUE)),
  sex            = sample(c("M", "F"), 50, TRUE),
  zip_codes      = sample(c("4556", "4557", "4558"), 50, TRUE),
  icd_codes      = sample(c("J06.9", "J18.9", "U07.1", "J44.1"), 50, TRUE),
  stringsAsFactors = FALSE
)

# Vaccination records (long: multiple rows per person)
vax_raw <- data.frame(
  patient_id       = rep(paste0("VAX", 1:40), each = 2),
  firstname        = rep(sample(c("Alice", "Bob", "Carol"), 40, TRUE), each = 2),
  last_name        = rep(sample(c("Smith", "Jones"), 40, TRUE), each = 2),
  birth_date       = rep(as.Date("1970-01-01") + sample(-5000:5000, 40, TRUE), each = 2),
  gender           = rep(sample(c("M", "F"), 40, TRUE), each = 2),
  postcode         = rep(sample(c("4556", "4557"), 40, TRUE), each = 2),
  medicare_number  = rep(paste0(sample(1000:9999, 40, TRUE), sample(10000:99999, 40, TRUE)),
                         each = 2),
  vaccine_delivered = c(rbind(rep("COVID-19 mRNA", 40), rep("COVID-19 mRNA", 40))),
  service_date     = c(rbind(
    as.Date("2024-01-01") + sample(0:180, 40, TRUE),
    as.Date("2024-06-01") + sample(0:180, 40, TRUE)
  )),
  stringsAsFactors = FALSE
)

## ----clean--------------------------------------------------------------------
df_cases <- clean_the_nest(
  cases_raw,
  data_type   = "cases",
  drop_eggs   = TRUE,
  id_var      = "identity",
  diagnosis   = "disease_name",
  lettername1 = "first_name",
  lettername2 = "surname",
  dob         = "date_of_birth",
  medicare    = "medicare_no",
  gender      = "gender",
  postcode    = "postcode",
  fn          = "indigenous_status",
  onset_date  = "date_of_onset"
)

df_hosp <- clean_the_nest(
  hosp_raw,
  data_type      = "hospital",
  drop_eggs      = TRUE,
  id_var         = "patient_id",
  lettername1    = "firstname",
  lettername2    = "last_name",
  dob            = "birth_date",
  medicare       = "medicare_number",
  gender         = "sex",
  postcode       = "zip_codes",
  icd_code       = "icd_codes",
  admission_date = "date_of_admission",
  discharge_date = "date_of_discharge"
)

df_vax <- clean_the_nest(
  vax_raw,
  data_type     = "vaccination",
  lie_nest_flat = TRUE,
  id_var        = "patient_id",
  lettername1   = "firstname",
  lettername2   = "last_name",
  dob           = "birth_date",
  medicare      = "medicare_number",
  gender        = "gender",
  postcode      = "postcode",
  vax_type      = "vaccine_delivered",
  vax_date      = "service_date"
)

# What does the cleaned case linelist look like?
head(df_cases[, c("lettername1", "lettername2", "dob", "age",
                   "onset_date", "diagnosis", "block1")], 4)

## ----preening-----------------------------------------------------------------
# Browse available schemes first
list_age_schemes(family = "surveillance", max_bands = 6)

## ----preening2----------------------------------------------------------------
# Apply the FluCAN sentinel scheme to the case linelist
df_cases <- preening(
  df_cases,
  age_col = "age",
  scheme  = "flucan_sentinel"
)

table(df_cases$age_group, useNA = "ifany")

## ----preening3----------------------------------------------------------------
# Or use filters instead of an exact name: narrow to broad national schemes
preening(
  df_cases,
  age_col  = "age",
  family   = "national_stats",
  focus    = "broad"
)$age_group |> table()

## ----roost--------------------------------------------------------------------
# Monthly case counts by pathogen
cases_monthly <- roost(
  df_cases,
  date_col   = "onset_date",
  time_unit  = "month",
  group_cols = "diagnosis"
)

cases_monthly

## ----roost-epiweek------------------------------------------------------------
# Epidemiological week counts (southern hemisphere default)
cases_epi <- roost(
  df_cases,
  date_col  = "onset_date",
  time_unit = "epiweek"
)

head(cases_epi)

## ----roost-season-------------------------------------------------------------
# Seasonal aggregation — useful for respiratory virus surveillance
cases_season <- roost(
  df_cases,
  date_col  = "onset_date",
  time_unit = "season_year"
)

cases_season

## ----corncrake----------------------------------------------------------------
factors <- data.frame(
  diagnosis    = rep(c("COVID-19", "Influenza A", "RSV"), each = 1),
  date_start   = as.Date("2024-01-01"),
  date_end     = as.Date(NA),   # open-ended: one factor for the whole series
  factor       = c(1.6, 2.1, 2.8),
  factor_lower = c(1.3, 1.7, 2.2),
  factor_upper = c(2.0, 2.6, 3.5),
  source       = "Illustrative multiplier, SCPHU surveillance evaluation 2025"
)

cases_corrected <- corncrake(
  cases_monthly,
  factor_table = factors,
  group_by     = "diagnosis"
)

cases_corrected[, c("diagnosis", "month", "n", "ascertainment_factor", "corrected_count")]

## ----molting------------------------------------------------------------------
result <- molting(df_cases)

# The de-identified dataset — no names, no DOB, no Medicare
names(result$deidentified)

# The lookup table — keep this separate and secure
head(result$lookup[, 1:3])

## ----homing-------------------------------------------------------------------
# Authorised relink — e.g. for clinical follow-up
relinked <- homing(
  deidentified_data = result$deidentified,
  lookup_table      = result$lookup
)

# Original identifiers are back
"lettername1" %in% names(relinked)

