# Age (per year) and being uninsured among adults with a history of cancer, NHANES 2013-2018.
# Wahab et al. (2022), Medicine, doi:10.1097/MD.0000000000030539.
# Headline: adjusted OR 0.95 (95% CI 0.93-0.96) per year of age, from the paper's one
# multivariable model of six sociodemographic predictors entered together (Table 3).

association <- list(
  id = "row275", row = 275, doi = "10.1097/MD.0000000000030539",
  cycles = c("2013-2014", "2015-2016", "2017-2018"),
  weight = "WTINT2YR",
  family = "logistic", term = "age",
  published = list(measure = "OR", estimate = 0.95, low = 0.93, high = 0.96, n = 1681, events = 85,
                   contrast = "per year of age (odds of being uninsured against insured)"),
  left_out = c(
    "annual household income in five bands (INDHHIN2): 2021-2023 releases no household income in dollars, only the family income-to-poverty ratio (INDFMPIR), a different measure, so the harmonized model leaves income out, and with it the paper's exclusion of those with income unknown",
    "years in the US among the foreign-born split at 40 (20-39 against 40 or more): 2021-2023 records length of time in the US only up to '20 years or more' (DMDYRUSR), so the harmonized model codes US born, foreign-born under 20 years, and 20 years or more"
  ),
  build = function(cycle) {
    demo <- demographics(cycle)
    dm <- component("DEMO", cycle)
    d <- merge_all(demo, component("HIQ", cycle, "HIQ011"), component("MCQ", cycle, "MCQ220"))
    from_demo <- function(v) if (has(dm, v)) dm[[v]][match(d$SEQN, dm$SEQN)] else rep(NA, nrow(d))
    # Covered by health insurance or any health care plan (HIQ011 1) or not (2); refused and
    # don't know are missing.
    d$uninsured <- ifelse(d$HIQ011 %in% 2, 1, ifelse(d$HIQ011 %in% 1, 0, NA))
    d$sex <- factor(d$sex, levels = 1:2, labels = c("male", "female"))
    # RIDRETH3: non-Hispanic White, Black, other (Asian 6, other or multiracial 7), Hispanic
    # (Mexican American 1, other Hispanic 2).
    race <- from_demo("RIDRETH3")
    d$race4 <- factor(ifelse(race %in% 3, "white", ifelse(race %in% 4, "black", ifelse(race %in% c(6, 7), "other",
                        ifelse(race %in% 1:2, "hispanic", NA)))), levels = c("white", "black", "other", "hispanic"))
    d$education4 <- factor(ifelse(d$education %in% 5, "college graduate", ifelse(d$education %in% 1:2, "less than high school",
                             ifelse(d$education %in% 3, "high school or GED", ifelse(d$education %in% 4, "some college", NA)))),
                           levels = c("college graduate", "less than high school", "high school or GED", "some college"))
    # Born in the US (DMDBORN4 1) or not (2), and the foreign-born's years in the US: DMDYRSUS
    # (2013-2018: 1-5 under 20 years, 6-7 20 to 39, 8-9 40 or more) or DMDYRUSR (2021-2023: 1-5
    # under 20 years, 6 20 or more). Refused or don't know is missing.
    born <- from_demo("DMDBORN4")
    detailed <- has(dm, "DMDYRSUS")
    years <- if (detailed) from_demo("DMDYRSUS") else from_demo("DMDYRUSR")
    foreign <- born %in% 2
    under_20 <- foreign & years %in% 1:5
    d$born4 <- factor(ifelse(born %in% 1, "us born", ifelse(under_20, "under 20 years",
                        ifelse(detailed & foreign & years %in% 6:7, "20-39 years",
                          ifelse(detailed & foreign & years %in% 8:9, "40 years or more", NA)))),
                      levels = c("us born", "under 20 years", "20-39 years", "40 years or more"))
    d$born3 <- factor(ifelse(born %in% 1, "us born", ifelse(under_20, "under 20 years",
                        ifelse(foreign & years %in% (if (detailed) 6:9 else 6), "20 years or more", NA))),
                      levels = c("us born", "under 20 years", "20 years or more"))
    # Annual household income (INDHHIN2, 2013-2018 only) in the paper's bands: under $20,000
    # (1-4 and 13, "under $20,000"), $20,000-44,999 (5-7 and 12, "$20,000 and over"),
    # $45,000-74,999 (8-10), $75,000-99,999 (14), $100,000 or more (15); refused and don't know
    # missing. The copy without code 12 leaves "$20,000 and over" missing (a variant).
    income <- from_demo("INDHHIN2")
    band <- ifelse(income %in% c(1:4, 13), "under 20k", ifelse(income %in% c(5:7, 12), "20-45k", ifelse(income %in% 8:10, "45-75k",
              ifelse(income %in% 14, "75-100k", ifelse(income %in% 15, "100k or more", NA)))))
    bands <- c("under 20k", "20-45k", "45-75k", "75-100k", "100k or more")
    d$income5 <- factor(band, levels = bands)
    d$income5_strict <- factor(ifelse(income %in% 12, NA, band), levels = bands)
    # Adults (MCQ220 is asked from 20) ever told they had cancer or a malignancy, with a yes or no
    # answer about health insurance.
    d$in_population <- d$age >= 20 & d$MCQ220 %in% 1 & !is.na(d$uninsured)
    d
  },
  formula = uninsured ~ age + sex + race4 + education4 + born4 + income5,
  formula_harmonized = uninsured ~ age + sex + race4 + education4 + born3,
  variants = list(
    list(label = "unweighted", weighted = FALSE),
    list(label = "age alone (published 0.94, 0.93-0.96)", formula = uninsured ~ age, sample = "own"),
    list(label = "age alone, unweighted", formula = uninsured ~ age, sample = "own", weighted = FALSE),
    list(label = "income code 12 ('$20,000 and over') missing", sample = "own",
         formula = uninsured ~ age + sex + race4 + education4 + born4 + income5_strict),
    list(label = "years in US in three groups (harmonized coding) only", sample = "own",
         formula = uninsured ~ age + sex + race4 + education4 + born3 + income5),
    list(label = "harmonized plus income-to-poverty ratio (continuous)", sample = "own",
         formula = uninsured ~ age + sex + race4 + education4 + born3 + pir)
  ),
  # What the coding check found this paper's analysis to have computed otherwise than its text
  # says (the plan's section 6.6).
  departures = list(),
  choices = list(
    list(choice = "population", decision = "adults 20 and older ever told they had cancer or a malignancy (MCQ220 1) who answered yes or no about health insurance (HIQ011 1 or 2)",
         reason = "as stated; it gives the flow chart's 1,684 and 1,681 (85 uninsured) and Table 1's weighted totals (25,982,352; 1,129,421 uninsured) exactly"),
    list(choice = "design", decision = "interview weight WTINT2YR divided by 3 with SDMVSTRA and SDMVPSU; the cancer population is a domain of the full design, so cases with a missing predictor stay in the variance estimate",
         reason = "the paper's weights and design, with SAS's NOMCAR option; whether it subset the data or used a domain is unstated, and here both give the same estimate and interval (45 design degrees of freedom). The published intervals are matched to the second decimal once SAS's default variance adjustment, (n-1)/(n-p) with n = 1,681, is applied, so the paper analyzed the 1,681 as a subset; the estimates are unchanged"),
    list(choice = "model sample", decision = "complete cases on the six predictors (1,567, 75 uninsured)",
         reason = "SAS fits the model on complete cases; the paper reports the missing counts (education 3, income 104, US birth and length of stay 10), and these codings match each exactly"),
    list(choice = "household income bands", decision = "INDHHIN2 1-4 and 13 under $20,000; 5-7 and 12 ('$20,000 and over') $20,000-44,999; 8-10 $45,000-74,999; 14 $75,000-99,999; 15 $100,000 or more; refused and don't know missing",
         reason = "unstated; only this grouping matches Table 1's counts (348, 526, 301, 123, 279, with 104 missing). Code 12 can't be placed in one band; leaving it missing gives 0.948 (variant)"),
    list(choice = "US birth and length of stay", decision = "US born (DMDBORN4 1); foreign-born by DMDYRSUS under 20 years (1-5), 20-39 (6-7), 40 or more (8-9); refused or don't know missing",
         reason = "the paper's groups; Table 1's counts (1,400, 56, 98, 117, with 10 missing) match exactly"),
    list(choice = "race, education, sex", decision = "RIDRETH3 non-Hispanic White, Black, other (Asian and other or multiracial), Hispanic (Mexican American and other Hispanic); DMDEDUC2 college graduate, less than high school (1-2), high school or GED, some college; male as reference",
         reason = "as stated; Table 1's counts match exactly"),
    list(choice = "age", decision = "years (RIDAGEYR, top-coded at 80), continuous, per year",
         reason = "the Results read the unadjusted estimate per year ('6% lower in 1-year older participants'); with these codings every odds ratio in Tables 2 and 3 is reproduced to the second decimal")
  )
)
