library(mudnester)

Why mudnester?

Surveillance data arrives messy — column names that differ between extracts, dates stored as character strings, vaccination records spread across one row per dose, identifiers that need stripping before sharing. mudnester handles all of this in a consistent, reproducible pipeline that feeds directly into record linkage (via starling::murmuration()) and visualisation (via bowerbird).

The package is named after the Corcoracidae — the White-winged Chough, which builds its mud nest in meticulous layers, reinforcing each layer before adding the next. That is exactly how a mudnester pipeline works: each function adds a layer, and none of them are trustworthy until the layer below is sound.

The pipeline at a glance

Raw surveillance data
(EDIS / iPM / AIR / NoCS / custom extracts)
         │
         ▼
clean_the_nest()   ─── standardise names, parse dates, derive blocking vars
         │
         ├── preening()      ─── age categorisation (~50 schemes)
         │
         ├── plumage()       ─── comorbidity detection (29 conditions, ICD-10-AM)
         │
         ├── roost()         ─── aggregate to time unit with zero-filling
         │        │
         │        └── corncrake()  ─── correct for under-ascertainment
         │
         ├── flyway()        ─── join several linked event dates into one table
         │        │              (onset, admission, ICU, death -- via roost() per event)
         │        └── corncrake()  ─── severity_anchor: CFR/IFR-anchor inversion
         │
         └── molting()       ─── de-identify with hash-based lookup
                   │
                   └── homing()   ─── relink when authorised

This vignette walks through a complete worked example. Each function has its own detailed reference vignette (see the links at the end of each section).


Synthetic data

We use three small synthetic datasets that mirror real SCPHU data structures.

set.seed(42)
n <- 80

# Case notifications (linelist style)
cases_raw <- data.frame(
  identity        = paste0("PT", seq_len(n)),
  first_name      = sample(c("James", "Sarah", "Michael", "Emma", "William"), n, TRUE),
  surname         = sample(c("Smith", "Jones", "Williams", "Taylor", "Brown"), n, TRUE),
  date_of_birth   = as.Date("1980-01-01") + sample(-10000:10000, n, TRUE),
  date_of_onset   = as.Date("2024-01-01") + sample(0:364, n, TRUE),
  disease_name    = sample(c("COVID-19", "Influenza A", "RSV"), n, TRUE),
  gender          = sample(c("M", "F"), n, TRUE),
  postcode        = sample(c("4556", "4557", "4558", "4560"), n, TRUE),
  medicare_no     = paste0(sample(1000:9999, n, TRUE), sample(10000:99999, n, TRUE)),
  indigenous_status = sample(c("Non-Indigenous", "Aboriginal", "Torres Strait Islander",
                                "Unknown"), n, TRUE, prob = c(0.85, 0.08, 0.02, 0.05)),
  stringsAsFactors = FALSE
)

# Hospital admissions
hosp_raw <- data.frame(
  patient_id     = paste0("UR", seq_len(50)),
  firstname      = sample(c("James", "Sarah", "Michael"), 50, TRUE),
  last_name      = sample(c("Smith", "Jones", "Williams"), 50, TRUE),
  birth_date     = as.Date("1950-01-01") + sample(-5000:5000, 50, TRUE),
  date_of_admission = as.Date("2024-01-01") + sample(0:364, 50, TRUE),
  date_of_discharge = as.Date("2024-01-01") + sample(1:400, 50, TRUE),
  medicare_number = paste0(sample(1000:9999, 50, TRUE), sample(10000:99999, 50, TRUE)),
  sex            = sample(c("M", "F"), 50, TRUE),
  zip_codes      = sample(c("4556", "4557", "4558"), 50, TRUE),
  icd_codes      = sample(c("J06.9", "J18.9", "U07.1", "J44.1"), 50, TRUE),
  stringsAsFactors = FALSE
)

# Vaccination records (long: multiple rows per person)
vax_raw <- data.frame(
  patient_id       = rep(paste0("VAX", 1:40), each = 2),
  firstname        = rep(sample(c("Alice", "Bob", "Carol"), 40, TRUE), each = 2),
  last_name        = rep(sample(c("Smith", "Jones"), 40, TRUE), each = 2),
  birth_date       = rep(as.Date("1970-01-01") + sample(-5000:5000, 40, TRUE), each = 2),
  gender           = rep(sample(c("M", "F"), 40, TRUE), each = 2),
  postcode         = rep(sample(c("4556", "4557"), 40, TRUE), each = 2),
  medicare_number  = rep(paste0(sample(1000:9999, 40, TRUE), sample(10000:99999, 40, TRUE)),
                         each = 2),
  vaccine_delivered = c(rbind(rep("COVID-19 mRNA", 40), rep("COVID-19 mRNA", 40))),
  service_date     = c(rbind(
    as.Date("2024-01-01") + sample(0:180, 40, TRUE),
    as.Date("2024-06-01") + sample(0:180, 40, TRUE)
  )),
  stringsAsFactors = FALSE
)

Step 1 — clean_the_nest(): standardise and prepare

clean_the_nest() is the entry point for all three data types. It renames columns to the mudnester internal schema, validates date formats, strips and lowercases name fields, derives blocking variables for record linkage, and optionally produces a lean drop_eggs = TRUE dataset ready for starling::murmuration().

df_cases <- clean_the_nest(
  cases_raw,
  data_type   = "cases",
  drop_eggs   = TRUE,
  id_var      = "identity",
  diagnosis   = "disease_name",
  lettername1 = "first_name",
  lettername2 = "surname",
  dob         = "date_of_birth",
  medicare    = "medicare_no",
  gender      = "gender",
  postcode    = "postcode",
  fn          = "indigenous_status",
  onset_date  = "date_of_onset"
)

df_hosp <- clean_the_nest(
  hosp_raw,
  data_type      = "hospital",
  drop_eggs      = TRUE,
  id_var         = "patient_id",
  lettername1    = "firstname",
  lettername2    = "last_name",
  dob            = "birth_date",
  medicare       = "medicare_number",
  gender         = "sex",
  postcode       = "zip_codes",
  icd_code       = "icd_codes",
  admission_date = "date_of_admission",
  discharge_date = "date_of_discharge"
)

df_vax <- clean_the_nest(
  vax_raw,
  data_type     = "vaccination",
  lie_nest_flat = TRUE,
  id_var        = "patient_id",
  lettername1   = "firstname",
  lettername2   = "last_name",
  dob           = "birth_date",
  medicare      = "medicare_number",
  gender        = "gender",
  postcode      = "postcode",
  vax_type      = "vaccine_delivered",
  vax_date      = "service_date"
)

# What does the cleaned case linelist look like?
head(df_cases[, c("lettername1", "lettername2", "dob", "age",
                   "onset_date", "diagnosis", "block1")], 4)
#>   lettername1 lettername2        dob  age onset_date   diagnosis      block1
#> 1       james       jones 1981-03-27 43.5 2024-10-09         RSV F 4560 1981
#> 2     william       jones 1979-06-17 44.7 2024-02-15 Influenza A F 4556 1979
#> 3       james    williams 1986-06-07 38.4 2024-10-13         RSV F 4556 1986
#> 4       james       brown 2001-01-06 23.9 2024-12-12    COVID-19 F 4560 2001

See also: vignette("clean-the-nest", package = "mudnester") for the full reference, including the birth-cohort pattern, keep_vars, and troubleshooting common data formats.


Step 2 — preening(): age categorisation

With cleaned data in hand, preening() assigns each record to an age band using one of ~50 named, referenced schemes — from ABS 5-year bands to ATAGI program-specific bands.

# Browse available schemes first
list_age_schemes(family = "surveillance", max_bands = 6)
#>                     scheme       family                          focus n_bands
#>               ed_syndromic surveillance             surveillance|broad       6
#>            flucan_sentinel surveillance surveillance|broad|national_au       5
#>  hospital_admitted_patient surveillance                   surveillance       6
#>              nors_outbreak surveillance             surveillance|broad       5
#>         notifiable_std_bbv surveillance      surveillance|fine_grained       6
#>             racf_aged_care surveillance         surveillance|aged_care       5
#>  age_range
#>         0+
#>         0+
#>         0+
#>         0+
#>         0+
#>         0+
# Apply the FluCAN sentinel scheme to the case linelist
df_cases <- preening(
  df_cases,
  age_col = "age",
  scheme  = "flucan_sentinel"
)

table(df_cases$age_group, useNA = "ifany")
#> 
#>   0-4  5-15 16-49 50-64   65+ 
#>     0     0    43    27    10
# Or use filters instead of an exact name: narrow to broad national schemes
preening(
  df_cases,
  age_col  = "age",
  family   = "national_stats",
  focus    = "broad"
)$age_group |> table()
#> 
#>     0  1-14 15-24 25-44 45-64   65+ 
#>     0     0    10    27    33    10

See also: vignette("preening", package = "mudnester") for the full scheme library, filter logic, and custom schemes.


Step 3 — roost(): aggregate to a time unit

roost() collapses individual records into counts at any standard surveillance time unit, with explicit zero-filling so epi curves never silently skip empty weeks.

# Monthly case counts by pathogen
cases_monthly <- roost(
  df_cases,
  date_col   = "onset_date",
  time_unit  = "month",
  group_cols = "diagnosis"
)

cases_monthly
#> # A tibble: 36 × 3
#>    diagnosis month          n
#>    <chr>     <date>     <int>
#>  1 COVID-19  2024-01-01     2
#>  2 COVID-19  2024-02-01     1
#>  3 COVID-19  2024-03-01     3
#>  4 COVID-19  2024-04-01     1
#>  5 COVID-19  2024-05-01     2
#>  6 COVID-19  2024-06-01     2
#>  7 COVID-19  2024-07-01     3
#>  8 COVID-19  2024-08-01     3
#>  9 COVID-19  2024-09-01     2
#> 10 COVID-19  2024-10-01     2
#> # ℹ 26 more rows
#> 
#> -- roost_meta --------------------------------------
#>  time_unit  : month 
#>  date_range : 2024-01-02 to 2024-12-29 
#>  group_cols : diagnosis 
#>  hemisphere : southern 
#>  n_rows_in  : 80
# Epidemiological week counts (southern hemisphere default)
cases_epi <- roost(
  df_cases,
  date_col  = "onset_date",
  time_unit = "epiweek"
)

head(cases_epi)
#> # A tibble: 6 × 3
#>   epiyear epiweek     n
#>     <dbl>   <dbl> <int>
#> 1    2024       1     1
#> 2    2024       2     0
#> 3    2024       3     2
#> 4    2024       4     0
#> 5    2024       5     3
#> 6    2024       6     0
#> 
#> -- roost_meta --------------------------------------
#>  time_unit  : epiweek 
#>  date_range : 2024-01-02 to 2024-12-29 
#>  hemisphere : southern 
#>  n_rows_in  : 80
# Seasonal aggregation — useful for respiratory virus surveillance
cases_season <- roost(
  df_cases,
  date_col  = "onset_date",
  time_unit = "season_year"
)

cases_season
#> # A tibble: 5 × 2
#>   season_year     n
#>   <chr>       <int>
#> 1 Autumn 2024    22
#> 2 Spring 2024    20
#> 3 Summer 2023     9
#> 4 Summer 2024     9
#> 5 Winter 2024    20
#> 
#> -- roost_meta --------------------------------------
#>  time_unit  : season_year 
#>  date_range : 2024-01-02 to 2024-12-29 
#>  hemisphere : southern 
#>  n_rows_in  : 80

See also: vignette("roost", package = "mudnester") for all time units, the biannual option, zero-filling behaviour, and pairing with bowerbird::roost_plot().


Step 4 — corncrake(): correct for under-ascertainment

Every notification system under-ascertains true disease burden to some degree, and that degree rarely stays constant — it typically varies by age group and over time. corncrake() applies a stratified, time-varying multiplier factor directly to roost() output, returning a corrected count alongside uncertainty bounds wherever they can be derived.

factors <- data.frame(
  diagnosis    = rep(c("COVID-19", "Influenza A", "RSV"), each = 1),
  date_start   = as.Date("2024-01-01"),
  date_end     = as.Date(NA),   # open-ended: one factor for the whole series
  factor       = c(1.6, 2.1, 2.8),
  factor_lower = c(1.3, 1.7, 2.2),
  factor_upper = c(2.0, 2.6, 3.5),
  source       = "Illustrative multiplier, SCPHU surveillance evaluation 2025"
)

cases_corrected <- corncrake(
  cases_monthly,
  factor_table = factors,
  group_by     = "diagnosis"
)

cases_corrected[, c("diagnosis", "month", "n", "ascertainment_factor", "corrected_count")]
#> # A tibble: 36 × 5
#>    diagnosis month          n ascertainment_factor corrected_count
#>    <chr>     <date>     <int>                <dbl>           <dbl>
#>  1 COVID-19  2024-01-01     2                  1.6             3.2
#>  2 COVID-19  2024-02-01     1                  1.6             1.6
#>  3 COVID-19  2024-03-01     3                  1.6             4.8
#>  4 COVID-19  2024-04-01     1                  1.6             1.6
#>  5 COVID-19  2024-05-01     2                  1.6             3.2
#>  6 COVID-19  2024-06-01     2                  1.6             3.2
#>  7 COVID-19  2024-07-01     3                  1.6             4.8
#>  8 COVID-19  2024-08-01     3                  1.6             4.8
#>  9 COVID-19  2024-09-01     2                  1.6             3.2
#> 10 COVID-19  2024-10-01     2                  1.6             3.2
#> # ℹ 26 more rows
#> 
#> -- roost_meta --------------------------------------
#>  time_unit  : month 
#>  date_range : 2024-01-02 to 2024-12-29 
#>  group_cols : diagnosis 
#>  hemisphere : southern 
#>  n_rows_in  : 80

See also: vignette("corncrake", package = "mudnester") for the ratio-estimate method, method = "severity_anchor" (the case-fatality/infection-fatality-rate anchor inversion – useful early in a novel outbreak before any seroprevalence survey exists), the two different uncertainty modes (ci_method), and rate calculation via denominator_col.

For method = "severity_anchor", both an observed case count and a severity outcome count (e.g. deaths) are needed together, at the same stratification and time. flyway() builds exactly that shape from a linked cohort’s several milestone dates in one call – see vignette("flyway", package = "mudnester").


Step 5 — molting() + homing(): de-identification

Before sharing aggregated or linelist outputs externally, molting() strips identifying columns and creates a secure lookup table. [homing()] reverses this when authorised relinking is needed.

result <- molting(df_cases)

# The de-identified dataset — no names, no DOB, no Medicare
names(result$deidentified)
#>  [1] "row_hash"   "age"        "onset_date" "block1"     "block2"    
#>  [6] "block3"     "id_var"     "diagnosis"  "postcode"   "gender"    
#> [11] "fn"         "age_group"

# The lookup table — keep this separate and secure
head(result$lookup[, 1:3])
#> # A tibble: 6 × 3
#>   row_hash                                               lettername1 lettername2
#>   <chr>                                                  <chr>       <chr>      
#> 1 40c153df7c293df4c75fea6ffe8cfd99a1fb693a55a34cb4585dc… james       jones      
#> 2 5991aa368f4e5532cebfee39f423e93519a424dbf49a4b667c589… william     jones      
#> 3 6be7085b9d9e3ce408d4821201bdbdb1056d978710162021036e7… james       williams   
#> 4 38bad43cf4fc747e0879bd73206c1e658baf38149e20813b07226… james       brown      
#> 5 830a57430d167616576fa6a94b645455a9f98f564999861e105d8… sarah       brown      
#> 6 032f7ba4c752b8d6a01f3d88b54da3c52c9da9f553e3962ee3600… emma        williams
# Authorised relink — e.g. for clinical follow-up
relinked <- homing(
  deidentified_data = result$deidentified,
  lookup_table      = result$lookup
)

# Original identifiers are back
"lettername1" %in% names(relinked)
#> [1] TRUE

See also: vignette("molting", package = "mudnester") for hash algorithm selection, collision handling, and storage security guidance.
vignette("homing", package = "mudnester") for the relink workflow and unmatched-record handling.


Complete pipeline summary

library(mudnester)

# 1. Clean
df_cases <- clean_the_nest(cases_raw, data_type = "cases", ...)
df_hosp  <- clean_the_nest(hosp_raw,  data_type = "hospital", ...)
df_vax   <- clean_the_nest(vax_raw,   data_type = "vaccination",
                            lie_nest_flat = TRUE, ...)

# 2. Link (via starling — not part of mudnester)
# c2v   <- starling::murmuration(df_cases, df_vax, ...)
# c2v2h <- starling::murmuration(c2v, df_hosp, ...)

# 3. Age-categorise
df_cases <- preening(df_cases, age_col = "age", scheme = "flucan_sentinel")

# 4. Comorbidity detection (hospital data only)
df_hosp <- plumage(df_hosp, icd_column = "icd_code")

# 5. Aggregate
cases_monthly <- roost(df_cases, date_col = "onset_date", time_unit = "month",
                        group_cols = c("age_group", "diagnosis"))

# 6. Correct for under-ascertainment
cases_corrected <- corncrake(cases_monthly, factor_table = ascertainment_factors,
                              group_by = "diagnosis")

# 7. De-identify before sharing
result <- molting(df_cases)
# Store result$lookup securely, share result$deidentified

# 8. Relink when authorised
relinked <- homing(result$deidentified, result$lookup)

Further reading

Vignette Function Purpose
vignette("clean-the-nest") clean_the_nest() Full parameter reference, data types, birth-cohort pattern
vignette("preening") preening(), list_age_schemes() All ~50 age schemes, filter logic, custom bands
vignette("age-schemes") list_age_schemes() Full catalogue with citations
vignette("plumage") plumage() Comorbidity detection, dual-code strategy, AR-DRG codes
vignette("roost") roost() All time units, zero-filling, seasonal options
vignette("flyway") flyway() Joining several linked event dates, feeding corncrake()’s severity_anchor
vignette("corncrake") corncrake() Ratio-estimate and severity-anchor methods, uncertainty modes, rate calculation
vignette("brood") brood() Vaccine coverage, cohort designs, interrupted time series
vignette("molting") molting() Hash algorithms, collision handling, security
vignette("homing") homing() Relink workflow, partial matches