# Neutrophil-to-HDL cholesterol ratio (NHR) and MASLD, NHANES 2017-March 2020.
# Lu et al. (2024), BMC Gastroenterol, doi:10.1186/s12876-024-03394-6.
# Headline: OR 1.20 (95% CI 1.09-1.31) per unit of NHR, Model 3 (Table 2).

association <- list(
  id = "row104", row = 104, doi = "10.1186/s12876-024-03394-6",
  cycles = "2017-2020",
  weight = "WTMEC2YR", blood_file = "CBC",
  family = "logistic", term = "nhr",
  published = list(measure = "OR", estimate = 1.20, low = 1.09, high = 1.31, n = 4761, events = 2123,
                   contrast = "per unit of NHR (neutrophils, 10^3 cells/uL, over HDL cholesterol, mmol/L)"),
  left_out = character(),
  build = function(cycle) {
    demo <- demographics(cycle)
    alq <- alcohol(cycle)
    bio <- biochemistry(cycle)
    d <- merge_all(demo, body_measures(cycle), elastography(cycle), hepatitis(cycle), blood_count(cycle),
                   hdl_cholesterol(cycle), total_cholesterol(cycle), hba1c(cycle), smoking(cycle),
                   cardiometabolic_criteria(cycle, demo), liver_cancer(cycle),
                   component("MCQ", cycle, "MCQ510E"), diabetes_status(cycle), fasting_glucose(cycle),
                   component("DIQ", cycle, c("DIQ160")), hypertension_status(cycle),
                   leisure_activity(cycle)[, c("SEQN", "sedentary_minutes")])
    d$grams <- alcohol_grams(alq)[match(d$SEQN, alq$SEQN)]
    d$alt <- bio$alt[match(d$SEQN, bio$SEQN)]
    d$albumin <- bio$albumin[match(d$SEQN, bio$SEQN)]
    d$creatinine_umol <- bio$creatinine[match(d$SEQN, bio$SEQN)] * 88.42
    d$uric_umol <- bio$uric_acid[match(d$SEQN, bio$SEQN)] * 59.48
    d$tc_mmol <- cholesterol_mmol(d$tc)
    d$nhr <- d$neutrophils / cholesterol_mmol(d$hdl)
    heavy <- (d$sex == 1 & d$grams > 30) | (d$sex == 2 & d$grams > 20)
    viral <- d$hbsag %in% 1 | d$hcv_antibody %in% 1 | d$hcv_rna %in% 1
    d$masld <- as.integer(d$cap >= 274 & d$cardiometabolic >= 1)
    d$race <- factor(d$race)
    d$sex <- factor(d$sex)
    d$education <- factor(d$education)
    d$pir3 <- cut(d$pir, c(-Inf, 1, 4, Inf), right = FALSE, labels = c("<1", "1-3.9", ">=4"))
    d$smoking <- factor(d$smoking)
    prediabetes <- d$DIQ160 %in% 1 | (d$hba1c >= 5.7) %in% TRUE | (d$glucose >= 100) %in% TRUE
    d$diabetes_grade <- factor(ifelse(d$diabetes %in% 1, "diabetes", ifelse(prediabetes, "prediabetes", ifelse(!is.na(d$diabetes), "normal", NA))),
                               levels = c("normal", "prediabetes", "diabetes"))
    d$hypertension <- factor(d$hypertension)
    d$sedentary_hours <- d$sedentary_minutes / 60
    d$in_population <- d$age >= 20 & !is.na(d$nhr) & d$elastography_status %in% 1 & !is.na(d$cap) & !is.na(d$stiffness) &
      !is.na(heavy) & !heavy & !viral & !(d$MCQ510E %in% 5) & !(d$liver_cancer %in% 1)
    d
  },
  variants = list(
    list(label = "unweighted", weighted = FALSE),
    list(label = "crude, weighted (published 1.48, 1.38-1.56)", formula = masld ~ nhr),
    list(label = "crude, unweighted", formula = masld ~ nhr, weighted = FALSE),
    list(label = "Model 2, weighted (published 1.49, 1.37-1.62)", formula = masld ~ nhr + age + race + education + sex + pir3),
    list(label = "Model 2, unweighted", formula = masld ~ nhr + age + race + education + sex + pir3, weighted = FALSE)
  ),
  # What the coding check found this paper's analysis to have computed otherwise than its text
  # says (the plan's section 6.6).
  departures = list(
    list(kind = "reporting", affects_headline = FALSE, followed = NA,
         detail = "The Results describe Model 3's estimate for the fourth NHR quartile against the first (OR 3.00, 2.10-4.30) as 'each unit increase in the NHR ratio was associated with a 3-fold increase in the odds of MASLD prevalence', but Table 2's Model 3 estimate per unit of NHR is 1.20."),
    list(kind = "reporting", affects_headline = FALSE, followed = NA,
         detail = "The Results give the NHR quartiles as 0.191-2.065, 2.067-2.993, 3.000-4.207, and 4.211-19.028, but Tables 2 and 3 label them 0.231-2.022, 2.023-2.948, 2.951-4.172, and 4.173-16.769."),
    list(kind = "reporting", affects_headline = FALSE, followed = NA,
         detail = "Table 1 has cells that contradict their own columns: the first quartile's prediabetes count is 1,110 where the column needs 110, the female total is 2,497 where its columns sum to 2,494, and the third quartile's MASLD shares (50.75% and 59.25%) and BMI shares (19.66%, 20.11%, 50.23%) sum to 110% and 90%.")
  ),
  choices = list(
    list(choice = "weights", decision = "examination weight (WTMECPRP; WTPH2YR in 2021-2023)", reason = "the paper says it weighted as NCHS directs; weighted and unweighted fits both reproduce the headline"),
    list(choice = "age", decision = "20 and older", reason = "Results describe participants aged over 20; no age step is listed"),
    list(choice = "alcohol grams a day", decision = "questionnaire: drinks per drinking day times drinking days a year over 365, 14 g a drink", reason = "unstated"),
    list(choice = "elastography completeness", decision = "exam status complete with CAP and stiffness recorded", reason = "unstated"),
    list(choice = "MASLD cardiometabolic criteria", decision = "2023 multisociety criteria with missing fasting values counted as not met", reason = "fasting glucose and triglycerides exist only for the fasting subsample"),
    list(choice = "hypertension", decision = "told, taking medication, or mean blood pressure 140/90 or more", reason = "unstated"),
    list(choice = "diabetes grade", decision = "diabetes (told, medication, HbA1c 6.5% or more, or fasting glucose 126 or more); prediabetes (told, HbA1c 5.7% or more, or fasting glucose 100 or more); otherwise normal", reason = "unstated"),
    list(choice = "covariate coding", decision = "age, BMI, sedentary hours, cholesterol, ALT, HbA1c, albumin, creatinine, uric acid continuous; income in the paper's three groups", reason = "unstated; the paper lists them as continuous measures")
  ),
  formula = masld ~ nhr + age + race + education + sex + pir3 + bmi + diabetes_grade + hypertension + smoking +
    sedentary_hours + tc_mmol + alt + hba1c + albumin + creatinine_umol + uric_umol
)
