clean_the_nest() doesclean_the_nest() is the foundational layer of the
mudnester pipeline. It:
lettername1, dob,
onset_date) so every downstream function knows what to
expectDate classblock1,
block2, block3) used by
starling::murmuration()lie_nest_flat = TRUENothing downstream is trustworthy until this layer is structurally sound. Think of it as laying the mud — subsequent functions can only build on what is solid here.
data_type is required. The three values map to different
source systems in a Queensland public health context.
data_type |
Typical source | Key mandatory dates |
|---|---|---|
"cases" |
NNDSS, NoCS, EDIS linelists | onset_date |
"hospital" |
iPM, HBCIS, admitted patient collections | admission_date |
"vaccination" |
Australian Immunisation Register (AIR), local registers | vax_date |
Age and age categories are derived for "cases" and
"hospital" when dob and
onset_date / admission_date are supplied. They
are not derived for "vaccination" (the AIR does not
reliably carry age at time of vaccination).
set.seed(1)
cases_raw <- data.frame(
identity = paste0("PT", 1:20),
first_name = sample(c("James","Sarah","Michael"), 20, TRUE),
surname = sample(c("Smith","Jones","Williams"), 20, TRUE),
date_of_birth = as.Date("1970-01-01") + sample(-5000:5000, 20),
date_of_onset = as.Date("2024-03-01") + sample(0:120, 20),
disease_name = sample(c("COVID-19","Influenza A","RSV"), 20, TRUE),
gender = sample(c("M","F"), 20, TRUE),
postcode = sample(c("4556","4557","4560"), 20, TRUE),
medicare_no = paste0(sample(2000:9999, 20), sample(10000:99999, 20)),
indigenous_status = sample(c("Non-Indigenous","Aboriginal","Unknown"), 20, TRUE,
prob = c(0.85, 0.10, 0.05)),
stringsAsFactors = FALSE
)
df_cases <- clean_the_nest(
cases_raw,
data_type = "cases",
drop_eggs = TRUE,
id_var = "identity",
diagnosis = "disease_name",
lettername1 = "first_name",
lettername2 = "surname",
dob = "date_of_birth",
medicare = "medicare_no",
gender = "gender",
postcode = "postcode",
fn = "indigenous_status",
onset_date = "date_of_onset"
)
head(df_cases[, c("lettername1","lettername2","dob","age","diagnosis","block1")], 4)
#> lettername1 lettername2 dob age diagnosis block1
#> 1 james williams 1965-10-29 58.7 COVID-19 M 4556 1965
#> 2 michael smith 1961-08-23 62.8 Influenza A F 4557 1961
#> 3 james smith 1963-05-17 61.0 COVID-19 M 4556 1963
#> 4 sarah smith 1960-07-01 63.9 COVID-19 M 4560 1960
Notice that lettername1 is lowercased,
punctuation-stripped, and truncated to the first name token only.
block1 is gender + postcode + birth_year —
used as the primary blocking variable in
starling::murmuration().
When admission_date and discharge_date are
both supplied, length of stay (los, in days) is derived
automatically. admission_outcome is a factor indicating
whether the person was actually admitted.
set.seed(2)
hosp_raw <- data.frame(
patient_id = paste0("UR", 1:15),
firstname = sample(c("James","Sarah","Michael"), 15, TRUE),
last_name = sample(c("Smith","Jones"), 15, TRUE),
birth_date = as.Date("1945-01-01") + sample(0:10000, 15),
date_of_admission = as.Date("2024-06-01") + sample(0:180, 15),
date_of_discharge = as.Date("2024-06-15") + sample(0:180, 15),
medicare_number = paste0(sample(2000:9999, 15), sample(10000:99999, 15)),
sex = sample(c("M","F"), 15, TRUE),
zip_codes = sample(c("4556","4557"), 15, TRUE),
icd_codes = sample(c("J06.9","U07.1","J44.1"), 15, TRUE),
stringsAsFactors = FALSE
)
df_hosp <- clean_the_nest(
hosp_raw,
data_type = "hospital",
drop_eggs = TRUE,
id_var = "patient_id",
lettername1 = "firstname",
lettername2 = "last_name",
dob = "birth_date",
medicare = "medicare_number",
gender = "sex",
postcode = "zip_codes",
icd_code = "icd_codes",
admission_date = "date_of_admission",
discharge_date = "date_of_discharge"
)
df_hosp[, c("los","admission_outcome","icd_code")] |> head(4)
#> los admission_outcome icd_code
#> 1 -18 Admission J44.1
#> 2 75 Admission J06.9
#> 3 156 Admission J44.1
#> 4 46 Admission U07.1
The Australian Immunisation Register exports one row per vaccination
event. lie_nest_flat = TRUE pivots this to one row per
person, with vax_date_1, vax_type_1,
vax_date_2, vax_type_2, etc.
set.seed(3)
vax_raw <- data.frame(
patient_id = rep(paste0("VAX", 1:10), each = 2),
firstname = rep(c("Alice","Bob","Carol","Dan","Eve",
"Frank","Grace","Henry","Iris","Jack"), each = 2),
last_name = rep(c("Smith","Jones","Williams","Taylor","Brown",
"White","Black","Green","Blue","Red"), each = 2),
birth_date = rep(as.Date("1980-01-01") + sample(-2000:2000, 10), each = 2),
gender = rep(sample(c("M","F"), 10, TRUE), each = 2),
postcode = rep(sample(c("4556","4557"), 10, TRUE), each = 2),
medicare_number = rep(paste0(sample(2000:9999, 10), sample(10000:99999, 10)), each = 2),
vaccine_delivered = rep(c("COVID-19 XBB.1.5","COVID-19 XBB.1.5"), 10),
service_date = c(rbind(
as.Date("2024-01-15") + sample(0:30, 10),
as.Date("2024-05-01") + sample(0:30, 10)
)),
stringsAsFactors = FALSE
)
df_vax <- clean_the_nest(
vax_raw,
data_type = "vaccination",
lie_nest_flat = TRUE,
id_var = "patient_id",
lettername1 = "firstname",
lettername2 = "last_name",
dob = "birth_date",
medicare = "medicare_number",
gender = "gender",
postcode = "postcode",
vax_type = "vaccine_delivered",
vax_date = "service_date"
)
df_vax[, c("id_var","vax_date_1","vax_type_1","vax_date_2","vax_type_2")] |> head(4)
#> # A tibble: 4 × 5
#> id_var vax_date_1 vax_type_1 vax_date_2 vax_type_2
#> <chr> <dbl> <chr> <dbl> <chr>
#> 1 VAX1 19737 COVID-19 XBB.1.5 19847 COVID-19 XBB.1.5
#> 2 VAX10 19746 COVID-19 XBB.1.5 19870 COVID-19 XBB.1.5
#> 3 VAX2 19745 COVID-19 XBB.1.5 19857 COVID-19 XBB.1.5
#> 4 VAX3 19753 COVID-19 XBB.1.5 19852 COVID-19 XBB.1.5
In birth-cohort vaccine effectiveness studies (e.g. nirsevimab or
Abrysvo effectiveness in infants), the date of birth is the
cohort entry date. Pass the same column name to both dob
and cohort_entry_date — no duplication needed.
df_cohort <- clean_the_nest(
birth_cohort_data,
data_type = "cases",
id_var = "baby_id",
lettername1 = "first_name",
lettername2 = "last_name",
dob = "babys_date_of_birth",
cohort_entry_date = "babys_date_of_birth", # same column — aliased internally
cohort_exit_date = "end_of_followup_date",
gender = "sex"
)
clean_the_nest() understands the full structure of
Australian Medicare numbers and produces semantically named output
columns for each component, rather than opaque digit-count names.
A complete Medicare number has up to 11 characters:
| Component | Digits | Column | Description |
|---|---|---|---|
| Account identifier | 1–8 | medicare08 |
Unique household/account number. First digit is always 2–6. |
| Checksum | 9 | medicare09 |
Mathematically derived from digits 1–8; used to validate the number.
clean_the_nest() validates this automatically. |
| Card issue number | 10 | medicare10 |
Increments each time a new card is issued (lost, expired, family member added). Never 0. |
| IRN | 11 | medicare_irn |
Individual Reference Number — the digit printed left of a person’s name. Primary cardholder = 1, partner = 2, children = 3, 4, etc. Up to 9 people share a card. |
Two additional columns are always produced:
medicare_clean (spaces removed, as supplied) and
medicare_valid (logical — TRUE if the checksum
is mathematically correct).
| Linkage purpose | Use |
|---|---|
| Person-level linkage (cases ↔︎ AIR) | medicare09 (stable across card reissues and family
members) |
| Card-specific linkage (e.g. AIR dose records) | medicare10 |
| Household-level blocking | medicare08 |
| Distinguishing individuals on the same card | medicare_irn |
mc_data <- data.frame(
patient_id = c("PT001", "PT002", "PT003"),
# PT001: valid 10-digit number (no IRN appended)
# PT002: valid 11-digit number (IRN = 1)
# PT003: invalid checksum (digit 9 is 7, should be 3)
mcare = c("2428778132", "24287781321", "2428778172"),
stringsAsFactors = FALSE
)
# suppressWarnings() here because PT003 has an invalid checksum —
# that's exactly what we want to demonstrate.
df_mc <- suppressWarnings(suppressMessages(
clean_the_nest(mc_data, data_type = "cases",
id_var = "patient_id", medicare = "mcare")
))
df_mc[, c("id_var", "medicare08", "medicare09", "medicare10",
"medicare_irn", "medicare_valid")]
#> id_var medicare08 medicare09 medicare10 medicare_irn medicare_valid
#> 1 PT001 24287781 242877813 2428778132 <NA> TRUE
#> 2 PT002 24287781 242877813 2428778132 1 TRUE
#> 3 PT003 24287781 242877817 2428778172 <NA> FALSE
Notice that PT003 has medicare_valid = FALSE —
clean_the_nest() warns you about invalid numbers before
linkage, because an invalid Medicare number will never match in
starling::murmuration(). In real data, investigate these
rows: a common cause is the IRN being stored as part of the 10-digit
number rather than separately.
drop_eggs argumentdrop_eggs = TRUE retains only the columns needed for
linkage and downstream analysis. This produces a lean, manageable
dataset for early-stage work. Use keep_vars to retain
additional columns you need alongside the defaults.
# Without drop_eggs — all original columns plus derived ones
df_full <- clean_the_nest(
cases_raw, data_type = "cases",
id_var = "identity", lettername1 = "first_name", lettername2 = "surname",
dob = "date_of_birth", onset_date = "date_of_onset"
)
ncol(df_full)
#> [1] 19
# With drop_eggs — only the linkage and analysis essentials
df_lean <- clean_the_nest(
cases_raw, data_type = "cases", drop_eggs = TRUE,
id_var = "identity", lettername1 = "first_name", lettername2 = "surname",
dob = "date_of_birth", onset_date = "date_of_onset",
keep_vars = "disease_name" # retain this one extra
)
ncol(df_lean)
#> [1] 15
names(df_lean)
#> [1] "lettername1" "lettername2"
#> [3] "lettername1_lettername2" "lettername1_lettername2_dob"
#> [5] "age" "medicare_no"
#> [7] "onset_date" "block1"
#> [9] "block2" "block3"
#> [11] "id_var" "dob"
#> [13] "postcode" "gender"
#> [15] "disease_name"
Once your data is cleaned:
preening() — assign age bands using
one of ~50 named schemes (see vignette("preening"))plumage() — detect chronic
comorbidities from ICD-10-AM coding (see
vignette("plumage")). For hospital datasets, this is
typically the next step after clean_the_nest() — the
standardised icd_code column produced here is the direct
input to plumage(icd_column = "icd_code").roost() — aggregate to a time unit for
an epi curve (see vignette("roost"))starling::murmuration() —
probabilistic record linkage across the cleaned datasetsmolting() — de-identify before sharing
(see vignette("molting"))