# Weight-adjusted-waist index (WWI) and non-alcoholic fatty liver disease (CAP of 274 dB/m or
# more), NHANES 2017-March 2020. Hu et al. (2023), Eur J Med Res, doi:10.1186/s40001-023-01205-4.
# Headline: OR 3.44 (95% CI 3.09-3.82) per unit of WWI, Model 3 (Table 2).

association <- list(
  id = "row100", row = 100, doi = "10.1186/s40001-023-01205-4",
  cycles = "2017-2020",
  weight = "WTMEC2YR", blood_file = "HSCRP",
  # The paper says it used the examination weight, but its crude, age-sex-race, and full-model
  # estimates are each reproduced only without weights (variants below), so the published
  # estimate comes from an unweighted analysis.
  weighted = FALSE,
  family = "logistic", term = "wwi",
  published = list(measure = "OR", estimate = 3.44, low = 3.09, high = 3.82, n = 6587, events = 2874,
                   contrast = "per unit of WWI (cm per square root of kg)"),
  left_out = c("serum ferritin (measured in 2021-2023 only in children 1-5 and females 12-49)",
               "blood transfusion (not asked in 2021-2023)"),
  build = function(cycle) {
    demo <- demographics(cycle)
    diet <- dietary_totals(cycle, c("KCAL", "SUGR", "TFAT", "MOIS", "ALCO"), days = "mean_or_one")
    q_diabetes <- component("DIQ", cycle, "DIQ010")
    q_bp <- component("BPQ", cycle, "BPQ020")
    mcq <- component("MCQ", cycle)
    d <- merge_all(demo, body_measures(cycle), elastography(cycle), hepatitis(cycle), smoking(cycle),
                   diet, crp(cycle), q_diabetes, q_bp, mcq[, c("SEQN", intersect(c("MCQ510E", "MCQ092"), names(mcq)))],
                   leisure_activity(cycle)[, c("SEQN", "sedentary_minutes")])
    if (cycle != REPLICATION_CYCLE) d <- merge_all(d, ferritin(cycle))
    d$wwi <- d$waist / sqrt(d$weight)
    d$nafld <- as.integer(d$cap >= 274)
    d$race3 <- factor(ifelse(d$race == 3, "white", ifelse(d$race == 4, "black", "other")), levels = c("white", "black", "other"))
    heavy_drinking <- (d$sex == 1 & d$ALCO > 30) | (d$sex == 2 & d$ALCO > 20)
    d$sex <- factor(d$sex)
    d$education <- with_unclear(ifelse(d$education %in% 1:3, "high school or less", ifelse(d$education %in% 4:5, "more than high school", NA)),
                                c("high school or less", "more than high school"))
    d$pir3 <- with_unclear(cut(d$pir, c(-Inf, 1.35, 3.45, Inf), right = FALSE, labels = c("<1.35", "1.35-3.45", ">3.45")),
                           c("<1.35", "1.35-3.45", ">3.45"))
    d$smoking <- factor(d$smoking)
    d$hypertension <- with_unclear(ifelse(d$BPQ020 %in% 1, "yes", ifelse(d$BPQ020 %in% 2, "no", NA)), c("no", "yes"))
    d$diabetes <- with_unclear(ifelse(d$DIQ010 %in% 1, "yes", ifelse(d$DIQ010 %in% 2, "no", NA)), c("no", "yes"))
    d$transfusion <- if ("MCQ092" %in% names(d)) with_unclear(ifelse(d$MCQ092 %in% 1, "yes", ifelse(d$MCQ092 %in% 2, "no", NA)), c("no", "yes")) else NA
    # Table 1 splits each dietary intake at these values, with missing as "Unclear".
    split <- function(x, at) with_unclear(ifelse(x < at, "low", ifelse(x >= at, "high", NA)), c("low", "high"))
    d$energy <- split(d$KCAL, 1861.5)
    d$sugar <- split(d$SUGR, 88.44)
    d$fat <- split(d$TFAT, 76.35)
    d$moisture <- split(d$MOIS, 2454.88)
    d$sedentary <- d$sedentary_minutes
    d$in_population <- d$age >= 20 & !is.na(d$wwi) & d$elastography_status %in% 1 & !is.na(d$cap) &
      !is.na(d$smoking) & d$DIQ010 %in% 1:3 &
      !(d$hbsag %in% 1) & !(d$hcv_antibody %in% 1) & !(d$hcv_rna %in% 1) & !(d$MCQ510E %in% 5) &
      !(heavy_drinking %in% TRUE) & !((d$stiffness_iqr_ratio > 30) %in% TRUE)
    d
  },
  variants = list(
    list(label = "weighted, as the paper says", weighted = TRUE),
    list(label = "crude, unweighted (published 2.53, 2.37-2.71)", formula = nafld ~ wwi),
    list(label = "crude, weighted", formula = nafld ~ wwi, weighted = TRUE),
    list(label = "age, sex, race, unweighted (published 3.28, 3.01-3.56)", formula = nafld ~ wwi + age + sex + race3),
    list(label = "age, sex, race, weighted", formula = nafld ~ wwi + age + sex + race3, weighted = TRUE)
  ),
  # What the coding check found this paper's analysis to have computed otherwise than its text
  # says (the plan's section 6.6).
  departures = list(
    list(kind = "weighting", affects_headline = TRUE, followed = TRUE, evidence = "data",
         detail = "The Methods say the examination weight WTMECPRP was used, but Table 2's Model 1, Model 2, and Model 3 estimates are reproduced only unweighted (Model 3: 3.50, 3.16-3.87 unweighted and 4.31, 3.54-5.25 weighted, against 3.44, 3.09-3.82; Model 1: 2.51 and 2.95 against 2.53, 2.37-2.71).")
  ),
  choices = list(
    list(choice = "weights", decision = "unweighted", reason = "only unweighted fits reproduce the published crude, Model 2, and Model 3 estimates; the text names WTMECPRP"),
    list(choice = "alcohol intake for the heavy-drinking exclusion", decision = "mean of the two 24-hour recalls, or the first day's where there is one", reason = "unstated; the dietary covariates use two-day means"),
    list(choice = "race groups", decision = "RIDRETH1: non-Hispanic White, non-Hispanic Black, all others", reason = "Table 1 shows White, Black, Other"),
    list(choice = "education", decision = "high school or less versus more, with missing as its own level", reason = "Table 1 categories, with 'Unclear'"),
    list(choice = "income to poverty ratio", decision = "Table 1 categories (<1.35, 1.35-3.45, >3.45) with missing as its own level", reason = "Table 1"),
    list(choice = "hypertension and diabetes", decision = "self-reported diagnosis (BPQ020, DIQ010), with missing or borderline as their own level", reason = "unstated; Table 1 shows yes, no, unclear"),
    list(choice = "dietary covariates", decision = "two-day means split at Table 1's cutpoints, with missing as its own level", reason = "Table 1"),
    list(choice = "activity intensity", decision = "sedentary minutes (PAD680), continuous", reason = "the covariate text says sedentary; Table 1's means fit PAD680"),
    list(choice = "missing CRP, ferritin, or sedentary minutes", decision = "complete case", reason = "unstated")
  ),
  formula = nafld ~ wwi + age + sex + race3 + height + education + pir3 + smoking + hypertension + diabetes +
    transfusion + energy + sugar + fat + moisture + crp + ferritin + sedentary,
  formula_harmonized = nafld ~ wwi + age + sex + race3 + height + education + pir3 + smoking + hypertension + diabetes +
    energy + sugar + fat + moisture + crp + sedentary
)
