[
 {
  "id": "row050",
  "rank": 7,
  "row": 50,
  "doi": "10.3389/fpsyt.2024.1433990",
  "pmcid": "PMC11442334",
  "title": "Association of diastolic and systolic blood pressure with depression: a cross-sectional study from NHANES 2005-2018",
  "authors": [
   "Zhang, Huifeng",
   "Xu, Ying",
   "Xu, Yaying"
  ],
  "year": 2024,
  "journal": "Frontiers in Psychiatry",
  "table_a": {
   "predictor": "Blood pressure",
   "condition": "Depression",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "Weighted logistic regression showed that after fully adjusting for depression-related risk factors, there was a significant positive correlation between per 10 mmHg increase in DBP and depression (OR: 1.06, 95% CI: 1.00-1.12, P=0.04); however, only on the left side of the inflection point, SBP tended to decrease the odds of depression (P =0.09).",
   "table_location": "Table 2 (Weighted multivariate Logistic regression analysis for the association of prevalent depression with SBP and DBP), row 'Continuous DBP (per 10 mmHg+)', column 'Model 4' (95%CI, P)",
   "table_quote": "| Continuous DBP (per 10 mmHg+) | 1.05(1.00,1.10) | 0.07 | 1.07(1.01,1.13) | 0.01 | 1.08(1.02,1.14) | 0.01 | 1.07(1.01,1.13) | 0.02 | 1.06(1.00,1.12) | 0.04 |",
   "measure": "OR",
   "estimate": 1.06,
   "ci_low": 1.0,
   "ci_high": 1.12,
   "p_value": "0.04",
   "exposure_contrast": "per 10 mmHg increase in DBP (continuous)",
   "model_label": "Model 4 (fully adjusted)",
   "covariates_in_this_model": [
    "age",
    "sex",
    "education attainment",
    "ethnicity",
    "marital status",
    "poverty-income ratio",
    "smoking",
    "drinking status",
    "BMI",
    "physical activity level",
    "total energy intake",
    "arthritis",
    "thyroid problems",
    "cancer",
    "diabetes",
    "liver diseases",
    "CHD",
    "CHF",
    "CKD",
    "stroke",
    "hyperlipidemia",
    "antihypertensive drug",
    "antidepressant"
   ],
   "n_analytic": 26581,
   "n_quote": "A total of 70,190 participants were involved in 7 cycles, with exclusions made for 33,793 missing PHQ-9, 954 missing blood pressure measurements, and 8,862 missing covariates. Ultimately, 26,581 eligible participants were included ( Figure 1 ).",
   "events": 2261
  },
  "cycles": [
   "2005-2006",
   "2007-2008",
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": "adults; no minimum age stated",
   "inclusion": "All NHANES 2005-2018 participants (70,190) with a PHQ-9, blood pressure measurements, and complete covariates.",
   "exclusions": [
    "missing PHQ-9 (n = 33,793)",
    "missing blood pressure measurements (n = 954)",
    "missing covariates (n = 8,862)"
   ],
   "quote": "A total of 70,190 participants were involved in 7 cycles, with exclusions made for 33,793 missing PHQ-9, 954 missing blood pressure measurements, and 8,862 missing covariates. Ultimately, 26,581 eligible participants were included ( Figure 1 )."
  },
  "exposure": {
   "definition": "Diastolic blood pressure (mmHg) from the examination, 3-4 auscultatory readings with a mercury sphygmomanometer. Average per the NHANES protocol the paper describes: one reading -> that reading; more than one -> the first reading is always excluded; two readings -> the second reading. Entered as a continuous variable scaled per 10 mmHg.",
   "nhanes_variables": [
    "BPXDI1",
    "BPXDI2",
    "BPXDI3",
    "BPXDI4"
   ],
   "nhanes_files": [
    "BPX"
   ],
   "transform": "per 10 mmHg (continuous, linear)",
   "categories": "Headline is continuous. Secondary categorical coding (Table 2): DBP <60 (ref), 60-79, 80-89, >=90 mmHg. Quote: 'If a nonlinear correlation is presented, according to previous studies, the classified DBP is defined with cut-off values of 60 mmHg, 80 mmHg, and 90 mmHg'",
   "quote": "During examinations at Mobile Examination Centers (MEC) and in-home visits, all eligible individuals undergo three to four blood pressure measurements using mercury sphygmomanometers. [...] The calculation of systolic and diastolic blood pressure does not represent traditional averages but is computed according to the following protocol: if only one blood pressure reading was obtained, that reading is the average. If there is more than one blood pressure reading, the first reading is always excluded from the average. If only two blood pressure readings were obtained, the second blood pressure reading is the average."
  },
  "outcome": {
   "definition": "Depression = PHQ-9 total score >= 10 (sum of the nine items, each 0-3, range 0-27).",
   "nhanes_variables": [
    "DPQ010",
    "DPQ020",
    "DPQ030",
    "DPQ040",
    "DPQ050",
    "DPQ060",
    "DPQ070",
    "DPQ080",
    "DPQ090"
   ],
   "nhanes_files": [
    "DPQ"
   ],
   "quote": "Depression was determined based on the Patient Health Questionnaire-9 (PHQ-9), which assessed depressive symptoms present in the past two weeks. The sum of the individual scores for each question (0-3 points) constituted the depression score, with a total score ranging from 0 to 27 points, where higher scores indicated more severe depression. According to previous studies, a PHQ-9 total score ≥10 was defined as depression (25–27)."
  },
  "covariates": [
   {
    "name": "age",
    "coding": "Covariates text says 'age (<40 years, ≥60 years)'; Table 1 shows <60 vs ≥60 years; coding in the model not stated",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L RIDAGEYR."
   },
   {
    "name": "sex",
    "coding": "male/female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L RIAGENDR."
   },
   {
    "name": "race",
    "coding": "Mexican American, non-Hispanic black, non-Hispanic white, other Hispanic, other race (including multiracial)",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Five groups match RIDRETH1 in DEMO_L."
   },
   {
    "name": "education level",
    "coding": "less than college, college and above",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Which DMDEDUC2 levels count as 'college and above' is not stated (Table 1 weighted share of 'College or higher' is 62.56%)."
   },
   {
    "name": "marital status",
    "coding": "divorced/separated/widowed, married/living with partner, never married",
    "nhanes_variables": [
     "DMDMARTL",
     "DMDMARTZ"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DMDMARTZ in DEMO_L has exactly these three groups (1 Married/Living with partner, 2 Widowed/Divorced/Separated, 3 Never married), released for ages 20+."
   },
   {
    "name": "poverty income ratio (PIR)",
    "coding": "<1.3, 1.3-3.5, >3.5 (Table 1: <1.3, 1.3–3.5, ≥3.5)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L INDFMPIR."
   },
   {
    "name": "alcohol consumption",
    "coding": "never, former, current (definitions not given)",
    "nhanes_variables": [
     "ALQ111",
     "ALQ121"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "Paper does not define the groups or name variables. 2021-2023 build: never = ALQ111 no; former = ALQ111 yes and ALQ121 = never in past 12 months; current = any drinking in past 12 months. Earlier cycles used different alcohol questions, so the groups may not match exactly."
   },
   {
    "name": "smoking",
    "coding": "never, former, current",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "Definitions not given; standard build: never = SMQ020 no; former = SMQ020 yes and SMQ040 not at all; current = SMQ040 every day or some days."
   },
   {
    "name": "total dietary energy intake",
    "coding": "high vs low, split at the median",
    "nhanes_variables": [
     "DR1TKCAL",
     "DR2TKCAL"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "Day 1 only or two-day mean not stated; median value and whether it is weighted not stated."
   },
   {
    "name": "physical activity level",
    "coding": "high vs low, split at the median metabolic equivalent; Table 1 also has an 'Unknown' group (6534, 20.11%)",
    "nhanes_variables": [
     "PAQ605",
     "PAQ610",
     "PAD615",
     "PAQ620",
     "PAQ625",
     "PAD630",
     "PAQ635",
     "PAQ640",
     "PAD645",
     "PAQ650",
     "PAQ655",
     "PAD660",
     "PAQ665",
     "PAQ670",
     "PAD675"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": false,
    "note": "Paper does not say which items or MET values; the 2007-2018 questionnaire covers work, transport and leisure (2005-2006 used a different questionnaire). PAQ_L has only leisure-time moderate and vigorous activity (PAD790Q/U, PAD800, PAD810Q/U, PAD820) and sedentary minutes, so a total MET can't be built the same way; a leisure-only MET split at its median is the closest substitute."
   },
   {
    "name": "body mass index (BMI)",
    "coding": "kg/m2 (Methods); Table 1 shows <25 vs ≥25; coding in the model not stated",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "BMX_L BMXBMI."
   },
   {
    "name": "coronary heart disease (CHD)",
    "coding": "yes/no (self-report presumed; not defined)",
    "nhanes_variables": [
     "MCQ160C"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "MCQ_L MCQ160c."
   },
   {
    "name": "congestive heart failure (CHF)",
    "coding": "yes/no (not defined)",
    "nhanes_variables": [
     "MCQ160B"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "MCQ_L MCQ160b."
   },
   {
    "name": "stroke",
    "coding": "yes/no (not defined)",
    "nhanes_variables": [
     "MCQ160F"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "MCQ_L MCQ160f."
   },
   {
    "name": "cancer",
    "coding": "yes/no (not defined)",
    "nhanes_variables": [
     "MCQ220"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "MCQ_L MCQ220."
   },
   {
    "name": "thyroid disease",
    "coding": "yes/no (Table 1 'Thyroid problem'; ever vs current not stated)",
    "nhanes_variables": [
     "MCQ160M",
     "MCQ170M"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "MCQ_L MCQ160m, MCQ170m."
   },
   {
    "name": "arthritis",
    "coding": "yes/no (not defined)",
    "nhanes_variables": [
     "MCQ160A"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "MCQ_L MCQ160a."
   },
   {
    "name": "liver disease",
    "coding": "yes/no (Table 1 'Liver problem'; ever vs current not stated)",
    "nhanes_variables": [
     "MCQ160L",
     "MCQ170L"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "MCQ_L MCQ160l, MCQ170l."
   },
   {
    "name": "diabetes",
    "coding": "yes/no (definition not given)",
    "nhanes_variables": [
     "DIQ010",
     "LBXGH",
     "LBXGLU",
     "DIQ050",
     "DIQ070"
    ],
    "nhanes_files": [
     "DIQ",
     "GHB",
     "GLU"
    ],
    "in_2021_2023": true,
    "note": "Self-report, HbA1c, fasting glucose, insulin and diabetic pills all exist in 2021-2023 (DIQ_L, GHB_L, GLU_L); which the paper used is not stated."
   },
   {
    "name": "chronic kidney disease (CKD)",
    "coding": "CKD with A2, G3a, and above (KDIGO: albumin-creatinine ratio >= 30 mg/g or eGFR < 60)",
    "nhanes_variables": [
     "URXUMA",
     "URXUCR",
     "URDACT",
     "LBXSCR"
    ],
    "nhanes_files": [
     "ALB_CR",
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "ALB_CR_L URDACT and BIOPRO_L LBXSCR exist; eGFR equation not stated."
   },
   {
    "name": "hyperlipidemia",
    "coding": "yes/no (definition not given; Table 1 prevalence 70.28%)",
    "nhanes_variables": [
     "LBXTC",
     "LBDHDD",
     "LBXTR",
     "LBDLDL",
     "BPQ080"
    ],
    "nhanes_files": [
     "TCHOL",
     "HDL",
     "TRIGLY",
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Lab lipids, BPQ080 and BPQ101D (cholesterol medication) exist in 2021-2023; if the paper used lipid-lowering drug names from the prescription file, that part can't be built (RXQ_RX_L has no drug names)."
   },
   {
    "name": "antihypertensive drug",
    "coding": "use in the past month (Results: 'the use of antidepressants or antihypertensive drugs in the past month')",
    "nhanes_variables": [
     "RXDDRUG",
     "RXDDRGID"
    ],
    "nhanes_files": [
     "RXQ_RX"
    ],
    "in_2021_2023": false,
    "note": "Source not stated; 'past month' points to the prescription medication file, whose drug names are not released in 2021-2023 (RXQ_RX_L has only RXQ033 and RXQ050). BPQ150 (taking high blood pressure medication) is a self-report proxy."
   },
   {
    "name": "antidepressant",
    "coding": "use in the past month",
    "nhanes_variables": [
     "RXDDRUG",
     "RXDDRGID"
    ],
    "nhanes_files": [
     "RXQ_RX"
    ],
    "in_2021_2023": false,
    "note": "Needs drug names or classes; RXQ_RX_L has only whether any prescription medicine was taken and how many."
   }
  ],
  "design": {
   "weights": "WTMEC2YR divided by 7 (paper: '1/7 * WTMEC2YR')",
   "strata_psu": "Complex design accounted for with the R survey package; strata and PSU variables (SDMVSTRA, SDMVPSU) not named",
   "quote": "In this study, the selected weight was the laboratory examination weight (1/7 * WTMEC2YR), as the key variables of this study were primarily obtained at the MEC. To account for the complex sampling methods involved in the NHANES database, [...] Weighted regression analysis was performed using the “survey” package",
   "software": "R 4.3.3 (survey, rms, segmented, car, mice)",
   "missing_data": "complete case (participants missing PHQ-9, BP, or covariates excluded), though Table 1 keeps physical activity 'Unknown' as a group; sensitivity analysis 2 used multiple imputation (mice) for covariates with <10% missing",
   "quote_missing": "with exclusions made for 33,793 missing PHQ-9, 954 missing blood pressure measurements, and 8,862 missing covariates [...] In addition, two sensitivity analyses were performed: 1) without considering sampling weights; 2) multiple imputation for covariates with missing proportions below 10%."
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "Weighted logistic regression was used to estimate the association between exposure and outcome risk as odds ratios (ORs) and 95% confidence intervals (CIs). Multiple models were constructed: crude model, unadjusted; Model 1, adjusted for sex, age, race, and education level; Model 2, further adjusted for marital status, poverty-income ratio, alcohol consumption, smoking, energy intake, physical activity level, and BMI; Model 3, further adjusted for other comorbidities, including cancer, thyroid disease, arthritis, liver disease, diabetes, and chronic kidney disease; Model 4 further adjusted for cardiovascular comorbidities and medication history, including coronary artery disease, congestive heart failure, stroke, antihypertensive drugs, and antidepressants."
  },
  "unstated": [
   "minimum age or definition of 'adults' (the flow from 70,190 has no age step)",
   "whether pregnant women were excluded",
   "handling of DBP readings of 0 mmHg or other implausible values",
   "whether the 2017-2018 auscultatory file (not the oscillometric one) was used (implied by 'mercury sphygmomanometers')",
   "coding of age in the models (continuous or <60/≥60; the covariates text reads 'age (<40 years, ≥60 years)')",
   "coding of BMI in the models (continuous or <25/≥25 as in Table 1)",
   "which DMDEDUC2 levels form 'less than college' vs 'college and above'",
   "definitions of never/former/current drinking and smoking",
   "physical activity: items, domains, MET values, the median cut, and how 'Unknown' entered the model",
   "energy intake: day 1 only or two-day mean, the median cut value, weighted or not",
   "definitions of diabetes and hyperlipidemia",
   "which questions define arthritis, thyroid disease, liver disease, cancer, CHD, CHF, stroke (ever told vs still have)",
   "eGFR equation for CKD",
   "how antihypertensive and antidepressant use were identified (drug file classes or questionnaire)",
   "strata and PSU variables (survey package used, not named)",
   "reference categories of categorical covariates",
   "which covariates' missingness produced the 8,862 exclusions",
   "handling of incomplete PHQ-9 item responses (refused, don't know, partial)",
   "variance estimation method and handling of single-PSU strata"
  ],
  "notes": "Quotes split with ' [...] ' are verbatim pieces; the covariates quote is split where the paper has an em dash ('other race [...] including multiracial'). Software is R with the survey package (not EmpowerStats). NHANES variable names in this file are our mapping; the paper names only WTMEC2YR. Methods and the Table 2 note assign covariates to models differently (Methods puts CKD in Model 3 and never lists hyperlipidemia; the Table 2 note puts CKD and hyperlipidemia in Model 4), but Model 4 contains the same 23 covariates either way, matching the Figure 2 and 3 legends and the Figure 4 legend's '22 risk factors other than stratification variables'. Table 2 rows '90 ≤DBP' print identical Model 1 and Model 2 results (1.34(1.02,1.76), P 0.04), possibly a copy error; the headline row is unaffected. The abstract's first estimate with a 95% CI is the DBP one; the SBP piecewise estimates (knot 129.7 mmHg) come later and the abstract gives them no CI. Table 1 keeps a physical activity 'Unknown' group (6534, 20.11%) even though participants missing covariates were excluded. The paper's BP text mentions in-home examinations for some participants. 2021-2023 BP is oscillometric only (BPXO_L Analytic Notes: auscultatory collection stopped after 2017-2018). No correction notice found in Europe PMC (no commentCorrectionList). No supplement in the PMC record.",
  "adjudication": null
 },
 {
  "id": "row100",
  "rank": 27,
  "row": 100,
  "doi": "10.1186/s40001-023-01205-4",
  "pmcid": "PMC10399060",
  "title": "Association of weight-adjusted-waist index with non-alcoholic fatty liver disease and liver fibrosis: a cross-sectional study based on NHANES",
  "authors": [
   "Hu, Qinggang",
   "Han, Kexing",
   "Shen, Jiapei",
   "Sun, Weijie",
   "Gao, Long",
   "Gao, Yufeng"
  ],
  "year": 2023,
  "journal": "European Journal of Medical Research",
  "table_a": {
   "predictor": "Weight-adjusted-waist index",
   "condition": "Metabolic-associated fatty liver conditions",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "In the model adjusted for all covariates, the effect values of WWI with NAFLD and liver fibrosis were (OR = 3.44, 95% CI: 3.09–3.82) and (OR = 2.40, 95% CI: 2.05–2.79), respectively.",
   "table_location": "Table 2 (Association of WWI with NAFLD and liver fibrosis), NAFLD block, row 'WWI', column 'Model 3, β (95% CI)' (values are ORs per the abstract and Results)",
   "table_quote": "| WWI | 2.53 (2.37, 2.71) | 3.28 (3.01, 3.56) | 3.44 (3.09, 3.82) |",
   "measure": "OR",
   "estimate": 3.44,
   "ci_low": 3.09,
   "ci_high": 3.82,
   "p_value": null,
   "exposure_contrast": "per 1-unit increase in WWI (cm/√kg), continuous",
   "model_label": "Model 3 (all covariates in Table 1 adjusted)",
   "covariates_in_this_model": [
    "Age(years)",
    "Gender",
    "Race",
    "Height (cm)",
    "Education",
    "PIR",
    "Smoking",
    "Hypertension",
    "Diabetes",
    "Blood transfusion",
    "Total daily energy intake (kcal)",
    "Total daily sugar intake (gm)",
    "Total daily fat intake (gm)",
    "Total daily moisture intake (gm)",
    "CRP (mg/l)",
    "Ferritin (ng/ml)",
    "Activity intensity (min)"
   ],
   "n_analytic": 6587,
   "n_quote": "Ultimately, the remaining 6587 participants were included in the study.",
   "events": 2874
  },
  "cycles": [
   "2017-March 2020 (pre-pandemic)"
  ],
  "population": {
   "age": ">=20",
   "inclusion": "NHANES 2017-March 2020 participants aged 20 or older with WWI and a completed liver ultrasound transient elastography (LUTE) exam, after the exclusions below",
   "exclusions": [
    "younger than 20 years (n = 6328)",
    "no WWI information (n = 1182)",
    "had not completed the LUTE test (n = 427)",
    "no smoking status (n = 3)",
    "no information on diabetes (n = 3)",
    "HBsAg positive (n = 41)",
    "hepatitis C antibody positive (n = 85)",
    "hepatitis C virus RNA positive (n = 81)",
    "autoimmune hepatitis (n = 10)",
    "males with alcohol intake > 30 g/d (n = 508)",
    "females with alcohol intake > 20 g/d (n = 428)",
    "unreliable liver stiffness, IQRe/median > 30% (n = 186)"
   ],
   "quote": "A total of 15,560 participants in the NHANES database were included in the survey. We excluded participants who were younger than 20 years of age (n = 6328), had no WWI information (n = 1182) and had not completed the LUTE test (n = 427). A small number of participants with no life information were also excluded, including no smoking status (n = 3) and no information on diabetes (n = 3). Participants with a history of viral hepatitis were also excluded, including those who were positive for hepatitis B surface antigen (HBsAg) (n = 41), positive for hepatitis C antibody (HCV-Ab) (n = 85) and positive for hepatitis C virus RNA (HCV-RNA) (n = 81). Participants with autoimmune hepatitis (AIH) were also excluded (n = 10). Participants defined as consuming large amounts of alcohol [18] were excluded for males (alcohol intake > 30 g/d) (n = 508) and females (alcohol intake > 20 g/d) (n = 428). In accordance with NHANES guidelines, results were unreliable for liver stiffness measures (LSM) at an interquartile range (IQRe)/median of > 30% and we also excluded this group of participants (n = 186). Ultimately, the remaining 6587 participants were included in the study."
  },
  "exposure": {
   "definition": "Weight-adjusted-waist index = waist circumference (cm) / sqrt(body weight (kg)), units cm/√kg, from MEC body measures",
   "nhanes_variables": [
    "BMXWAIST",
    "BMXWT"
   ],
   "nhanes_files": [
    "BMX"
   ],
   "transform": "none (continuous, per 1 unit); quartiles in secondary models",
   "categories": "Headline uses continuous WWI. Secondary models use quartiles (labelled 'Quintiles of WWI' in Table 2 but four groups): | Q1(8.443–10.534) | Reference | Reference | Reference |; Q2(10.535–11.109); Q3(11.110–11.697); Q4(11.698–14.137). Table 1 splits at the median: We grouped the participants based on the median WWI (11.11) for Group 1 (< 11.11) and Group 2 (> 11.11).",
   "quote": "WWI was calculated by dividing waist circumference in centimetres by the square root of body weight in kilograms [22]."
  },
  "outcome": {
   "definition": "NAFLD (hepatic steatosis) = median controlled attenuation parameter (CAP) >= 274 dB/m on FibroScan transient elastography",
   "nhanes_variables": [
    "LUXCAPM"
   ],
   "nhanes_files": [
    "LUX"
   ],
   "quote": "A median value of ≥ 274 dB/m for CAP was considered to be a marker of steatosis according to a study conducted by Eddowes et al. [19]."
  },
  "covariates": [
   {
    "name": "age",
    "coding": "years; Table 1 shows it continuous and also stratified (< 40, 40–59, > 60); coding in Model 3 not stated",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L RIDAGEYR."
   },
   {
    "name": "gender",
    "coding": "male, female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L RIAGENDR."
   },
   {
    "name": "race",
    "coding": "White, Black and Other races",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L RIDRETH1 or RIDRETH3; which variable and how Hispanic and Asian participants map to 'Other races' is not stated."
   },
   {
    "name": "education",
    "coding": "less than high school and high school, greater than high school (Table 1 adds 'Unclear')",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L DMDEDUC2."
   },
   {
    "name": "family income to poverty ratio (PIR)",
    "coding": "Table 1 shows categories < 1.35, 1.35–3.45, > 3.45, Unclear; coding in Model 3 not stated",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L INDFMPIR."
   },
   {
    "name": "height (cm)",
    "coding": "continuous",
    "nhanes_variables": [
     "BMXHT"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "BMX_L BMXHT."
   },
   {
    "name": "C-reactive protein (CRP, mg/l)",
    "coding": "continuous",
    "nhanes_variables": [
     "LBXHSCRP"
    ],
    "nhanes_files": [
     "HSCRP"
    ],
    "in_2021_2023": true,
    "note": "HSCRP_L LBXHSCRP (hs-CRP, mg/L)."
   },
   {
    "name": "serum ferritin (ng/ml)",
    "coding": "continuous",
    "nhanes_variables": [
     "LBXFER"
    ],
    "nhanes_files": [
     "FERTIN"
    ],
    "in_2021_2023": false,
    "note": "FERTIN_L measures ferritin only in ages 1-5 and females 12-49, so not available for the adult study population."
   },
   {
    "name": "smoking",
    "coding": "ever, now, never: at least 100 cigarettes in life, split by current smoking",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "SMQ_L SMQ020, SMQ040."
   },
   {
    "name": "hypertension",
    "coding": "yes, no, unclear (Table 1); definition not stated",
    "nhanes_variables": [
     "BPQ020"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Definition unstated. If self-reported diagnosis, BPQ_L BPQ020 (medication BPQ150); if measured, BPXO_L oscillometric readings."
   },
   {
    "name": "diabetes",
    "coding": "yes, no, unclear (Table 1); definition not stated",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "Definition unstated. If self-reported diagnosis, DIQ_L DIQ010 (GHB_L, GLU_L available for lab-based definitions)."
   },
   {
    "name": "blood transfusion",
    "coding": "yes, no, unclear (Table 1)",
    "nhanes_variables": [
     "MCQ092"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": false,
    "note": "MCQ_L (2021-2023) has no blood transfusion question."
   },
   {
    "name": "sedentary ('Activity intensity (min)' in Table 1)",
    "coding": "minutes, continuous; the variable is not named",
    "nhanes_variables": [
     "PAD680"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": true,
    "note": "The covariate text says 'sedentary'; Table 1 shows 'Activity intensity (min)' with mean 324.85 and 329.10 min, which is consistent with PAD680 (minutes sedentary activity) but the paper does not say. PAQ_L PAD680 exists."
   },
   {
    "name": "energy intake (kcal)",
    "coding": "mean of day 1 and day 2 recalls; Table 1 shows < 1861.5, ≥ 1861.5, Unclear",
    "nhanes_variables": [
     "DR1TKCAL",
     "DR2TKCAL"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "DR1TOT_L, DR2TOT_L."
   },
   {
    "name": "sugar intake (gm)",
    "coding": "mean of day 1 and day 2; Table 1 shows < 88.44, ≥ 88.44, Unclear",
    "nhanes_variables": [
     "DR1TSUGR",
     "DR2TSUGR"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "DR1TOT_L, DR2TOT_L."
   },
   {
    "name": "fat intake (gm)",
    "coding": "mean of day 1 and day 2; Table 1 shows < 76.35, ≥ 76.35, Unclear",
    "nhanes_variables": [
     "DR1TTFAT",
     "DR2TTFAT"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "DR1TOT_L, DR2TOT_L."
   },
   {
    "name": "water intake ('Total daily moisture intake (gm)' in Table 1)",
    "coding": "mean of day 1 and day 2; Table 1 shows < 2454.88, ≥ 2454.88, Unclear",
    "nhanes_variables": [
     "DR1TMOIS",
     "DR2TMOIS"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "DR1TOT_L, DR2TOT_L (moisture)."
   }
  ],
  "design": {
   "weights": "WTMECPRP (2017-March 2020 pre-pandemic MEC examination weight); for 2021-2023 the analogue is WTMEC2YR",
   "strata_psu": "not stated (only the weight is named)",
   "quote": "In accordance with NHANES guidelines, NHANES check sample weights were applied to the analysis of the LUTE data. Therefore, the special examination sample weights (Variable Name: WTMECPRP) for the 2017–2020.03 cycle were used in this study.",
   "software": "R 4.1.2 and EmpowerStats (Data collation and statistical analysis were done via R (4.1.2) and Empower Stats.)",
   "missing_data": "exclusion of missing exposure, outcome, smoking and diabetes; Table 1 carries 'Unclear' levels for education, PIR, hypertension, diabetes, blood transfusion and the four dietary covariates, which suggests a missing-indicator category in Model 3, but the paper does not say; handling of missing CRP, ferritin and activity is unstated",
   "quote_missing": "A small number of participants with no life information were also excluded, including no smoking status (n = 3) and no information on diabetes (n = 3)."
  },
  "model": {
   "family": "logistic (reported as OR for binary NAFLD; the Methods say 'multiple linear regression')",
   "weighted": true,
   "quote": "Multiple linear regression analysis was used to examine the relationship between the independent and dependent variables. A total of three models were generated based on the adjustment of covariates. Model 1: No adjustment for covariates. Model 2: Adjusted for age, race and gender. Model 3: All covariates in Table 1 are adjusted."
  },
  "unstated": [
   "regression family: the Methods say 'multiple linear regression' while every estimate is reported as an OR (logistic presumably)",
   "whether SDMVSTRA and SDMVPSU were used (only WTMECPRP is named)",
   "source of alcohol g/d for the heavy-drinking exclusion (24-hour recall day 1, two-day mean, or ALQ questionnaire) and whether participants without that information were kept",
   "definition of hypertension (self-report, medication, or measured BP)",
   "definition of diabetes and of its 'Unclear' level (borderline? missing?)",
   "whether age, PIR and the four dietary covariates enter Model 3 continuously or in the Table 1 categories",
   "how 'Unclear' (missing) categories enter the model",
   "which race variable (RIDRETH1 or RIDRETH3) and how Mexican American, other Hispanic and Asian participants are grouped into White, Black, Other",
   "what 'sedentary' / 'Activity intensity (min)' is (variable and units)",
   "the blood transfusion question used",
   "handling of missing CRP, ferritin and activity values",
   "what 'completed the LUTE test' means (exam status code) and whether any CAP reliability rule was applied beyond the stiffness IQR/median > 30% rule",
   "how autoimmune hepatitis was identified (MCQ510e presumably)",
   "whether participants with only a day-1 dietary recall were kept",
   "reference categories for categorical covariates"
  ],
  "notes": "Analysis in R 4.1.2 and EmpowerStats. Table 2 column heads say 'β (95% CI)' but the values are ORs (abstract, Results); its 'Quintiles of WWI' rows are quartiles (four groups), and two Model 2 cells carry an inline '< 0.0001'. Flow arithmetic does not close: 15,560 minus the listed exclusions (6328 + 1182 + 427 + 3 + 3 + 41 + 85 + 81 + 10 + 508 + 428 + 186 = 9282) leaves 6278, not the stated 6587 (Fig. 1 not available as text). The abstract gives the nonlinearity test as LLR < 0.01; Results and Table 3 give < 0.001. Table 3's 'Linear effect model 3.44 (3.09, 3.82)' equals Table 2 Model 3, so the threshold model is fully adjusted. The Methods say continuous variables are shown as means ± standard errors, but Table 1's note says mean ± SD. The stiffness IQR/median > 30% exclusion was applied to the whole sample, including the NAFLD analysis. The paper's second outcome, liver fibrosis (LSM ≥ 8.0 kPa), has whole-population Model 3 OR 2.40 (2.05, 2.79); it is not the headline for this Table A condition. Ferritin is reported for both WWI groups in Table 1 (means 153.57 and 157.61 ng/ml), so it was available for adults in 2017-2020 but is not in 2021-2023. For 2021-2023 use the MEC exam weight (LUX_L documentation: use examination weights for elastography unless merged with a more restrictive subsample).",
  "adjudication": null
 },
 {
  "id": "row106",
  "rank": 28,
  "row": 106,
  "doi": "10.1265/ehpm.24-00050",
  "pmcid": "PMC11211073",
  "title": "Association between blood cadmium and depression varies by age and smoking status in US adult women: a cross-sectional study from NHANES 2005–2016",
  "authors": [
   "Ji, Yewei",
   "Wang, Jinmin"
  ],
  "year": 2024,
  "journal": "Environmental Health and Preventive Medicine",
  "table_a": {
   "predictor": "Blood cadmium levels",
   "condition": "Depression",
   "population": "US adult females"
  },
  "headline": {
   "abstract_quote": "In the fully adjusted model, each incremental unit of blood cadmium was associated with a 33% rise in the prevalence of depression (OR = 1.33, 95% CI: 1.21–1.45).",
   "table_location": "Table 2 (Association between ln transform blood cadmium and PHQ-9 Score and Depression), block 'Fully adjusted model (Model3)', row 'Continuous', column 'Depression, OR (95% CI)'",
   "table_quote": "| Continuous | 0.44 (0.22, 0.65) | 1.33 (1.21, 1.45) |",
   "measure": "OR",
   "estimate": 1.33,
   "ci_low": 1.21,
   "ci_high": 1.45,
   "p_value": null,
   "exposure_contrast": "per 1-unit increase in ln-transformed blood cadmium (natural log of ug/L); the abstract says 'each incremental unit of blood cadmium', Table 2 and the Results specify the ln transform",
   "model_label": "Fully adjusted model (Model3)",
   "covariates_in_this_model": [
    "age",
    "race",
    "PIR",
    "BMI",
    "education level",
    "smoking status",
    "marital status",
    "hypertension",
    "diabetes"
   ],
   "n_analytic": 10868,
   "n_quote": "a total of 10,868 participants were included in our study (Fig. 1).",
   "events": 1173
  },
  "cycles": [
   "2005-2006",
   "2007-2008",
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016"
  ],
  "population": {
   "age": ">=20",
   "inclusion": "Women aged 20 or older in NHANES 2005-2016 with PHQ-9 depression data, blood cadmium, lead and mercury, and complete covariates",
   "exclusions": [
    "missing data on depression (n = 29,745)",
    "missing data on blood cadmium, lead and mercury concentrations (n = 6,265)",
    "males (n = 12,309)",
    "younger than 20 years old (n = 723)",
    "missing data on other covariates (n = 1,026)"
   ],
   "quote": "Initially, 60,936 participants were recruited, and, following the exclusion of those with missing data on depression (n = 29,745), missing data on blood cadmium, lead and mercury concentrations (n = 6,265), males (n = 12,309), individuals younger than 20 years old (n = 723), and participants with missing data on other covariates (n = 1,026), a total of 10,868 participants were included in our study (Fig. 1)."
  },
  "exposure": {
   "definition": "Whole-blood cadmium (ug/L) measured by inductively coupled plasma mass spectrometry, natural-log transformed ('ln transform blood cadmium')",
   "nhanes_variables": [
    "LBXBCD"
   ],
   "nhanes_files": [
    "PBCD"
   ],
   "transform": "ln (natural log), per 1 unit; quartiles in secondary models",
   "categories": "Headline uses continuous ln blood cadmium. Secondary models use quartiles: Blood cadmium levels (µg/L) spanned 0.07–0.22, 0.23–0.35, 0.36–0.59, and 0.6–10.8 for quartiles 1–4, respectively.",
   "quote": "Blood cadmium (BCd) was designed as the exposure variable. Blood cadmium is determined using inductively coupled plasma mass spectrometry. [...] Associations between ln transform blood cadmium (in quartiles) and depression and PHQ-9 score in different models were investigated using logistic regression and multiple linear regression."
  },
  "outcome": {
   "definition": "Depression = PHQ-9 total score (sum of nine items, each 0-3) >= 10",
   "nhanes_variables": [
    "DPQ010",
    "DPQ020",
    "DPQ030",
    "DPQ040",
    "DPQ050",
    "DPQ060",
    "DPQ070",
    "DPQ080",
    "DPQ090"
   ],
   "nhanes_files": [
    "DPQ"
   ],
   "quote": "Each item is evaluated on a 4-point ordinal scale indicating the prevalence of the symptom (0, not at all; 1, several days; 2, more than half the days; 3, nearly every day). The PHQ-9 total score is the sum of the scores from the nine items. Therefore, participants with a score of 10 or higher were defined as having depressive symptoms."
  },
  "covariates": [
   {
    "name": "age",
    "coding": "years, continuous (Table 1 mean ± SD)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L RIDAGEYR."
   },
   {
    "name": "race",
    "coding": "Mexican American / other Hispanic / non-Hispanic White / non-Hispanic Black / other races",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L RIDRETH1 (five levels as listed)."
   },
   {
    "name": "body mass index (BMI)",
    "coding": "text: stratified into < 25, 25–29.9, ≥ 30 kg/m2; Table 1 reports it as continuous; coding in Model 3 not stated",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "BMX_L BMXBMI."
   },
   {
    "name": "family income to poverty ratio (PIR)",
    "coding": "continuous in Table 1; coding in the model not stated",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L INDFMPIR."
   },
   {
    "name": "education level",
    "coding": "Less than high school / High school or GED / Above high school (Table 1)",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L DMDEDUC2; mapping of its five codes to three levels not stated."
   },
   {
    "name": "smoking status",
    "coding": "yes/no: smoked >= 100 cigarettes in life",
    "nhanes_variables": [
     "SMQ020"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "SMQ_L SMQ020."
   },
   {
    "name": "marital status",
    "coding": "yes/no; mapping not stated",
    "nhanes_variables": [
     "DMDMARTL"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "2021-2023 has DMDMARTZ (married/living with partner; widowed/divorced/separated; never married) instead of DMDMARTL; a yes/no split is buildable, but whether 'yes' included living with a partner is not stated."
   },
   {
    "name": "hypertension",
    "coding": "self-reported physician diagnosis, yes/no",
    "nhanes_variables": [
     "BPQ020"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "BPQ_L BPQ020."
   },
   {
    "name": "diabetes",
    "coding": "self-reported physician diagnosis; text says yes vs no, Table 1 shows Yes / No / Borderline",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "DIQ_L DIQ010 (1 yes, 2 no, 3 borderline)."
   }
  ],
  "design": {
   "weights": "not named",
   "strata_psu": "the paper says it adjusted for the complex multistage cluster survey design; strata and PSU variables not named",
   "quote": "All statistical analyses followed the Centers for Disease Control and Prevention (CDC) guidelines and adjusted for the complex multistage cluster survey design during analysis.",
   "software": "R 4.3.1 or EmpowerStats 2.0 (All analyses were performed with R (version 4.3.1) or Empowerstats (version 2.0).)",
   "missing_data": "complete case",
   "quote_missing": "participants with missing data on other covariates (n = 1,026)"
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "Associations between ln transform blood cadmium (in quartiles) and depression and PHQ-9 score in different models were investigated using logistic regression and multiple linear regression. [...] In Model 1, no covariates were adjusted. Model 2 included adjustments for age and race. Model 3 was adjusted for age, race, BMI, PIR, education level, marital status, smoking status, hypertension, and diabetes."
  },
  "unstated": [
   "which survey weight (none is named) and how it was combined across the six cycles",
   "whether SDMVSTRA and SDMVPSU were used (the text only says the design was accounted for)",
   "handling of blood cadmium values below the limit of detection",
   "how 'missing data on depression' was defined (any PHQ-9 item missing; refused or don't know codes 7 and 9)",
   "BMI coding in Model 3 (continuous as in Table 1 or the three stated categories)",
   "PIR coding in the model",
   "which marital status codes count as 'yes'",
   "diabetes coding in the model (yes/no per the text, or Yes/No/Borderline per Table 1)",
   "mapping of DMDEDUC2 codes to the three education levels",
   "reference categories for categorical covariates",
   "whether pregnant women were excluded (not mentioned)"
  ],
  "notes": "Analysis in R 4.3.1 or EmpowerStats 2.0. The abstract says 'each incremental unit of blood cadmium', but the estimate is per unit of ln blood cadmium (Table 2 title and Results). Gender is listed among covariates but the sample is all women and the Model 3 list omits it. Inclusion required non-missing blood lead and mercury as well as cadmium. The flow arithmetic closes: 60,936 - 29,745 - 6,265 - 12,309 - 723 - 1,026 = 10,868. Table 3's note lists a different covariate set for the subgroup models (age, race, BMI, marital status, smoking status, diabetes and hypertension; no PIR or education). The supplement (Additional file 1, Table S1) covers blood lead and mercury only and was not needed. In 2021-2023 PBCD_L carries the phlebotomy weight WTPH2YR (adults 18+: about 95% of examined gave blood, per the PBCD_L documentation), and PHQ-9 moved to ACASI.",
  "adjudication": null
 },
 {
  "id": "row121",
  "rank": 34,
  "row": 121,
  "doi": "10.1016/j.pmedr.2023.102306",
  "pmcid": "PMC10336672",
  "title": "The association between visceral adiposity index and chronic kidney disease in the elderly: A cross-sectional analysis of NHANES 2011–2018",
  "authors": [
   "Peng, Wei",
   "Han, Min",
   "Xu, Gang"
  ],
  "year": 2023,
  "journal": "Preventive Medicine Reports",
  "table_a": {
   "predictor": "Visceral adiposity index",
   "condition": "Chronic kidney disease",
   "population": "US middle-aged and older adults"
  },
  "headline": {
   "abstract_quote": "After fully adjusting for confounding factors, higher lnVAI was associated with a higher risk of CKD (OR, 1.23; 95 %CI, 1.02, 1.48).",
   "table_location": "Table 2, column 'Model 3' (OR (95% CI) and P), row 'lnVAI'",
   "table_quote": "| lnVAI | 1.35(1.20, 1.52) | <0.001 | 1.42(1.26, 1.61) | <0.001 | 1.23(1.02, 1.48) | 0.041 |",
   "measure": "OR",
   "estimate": 1.23,
   "ci_low": 1.02,
   "ci_high": 1.48,
   "p_value": "0.041",
   "exposure_contrast": "per 1-unit increase in lnVAI (natural log of VAI), i.e. per e-fold (about 2.72 times) higher VAI: 'The adjusted OR (95 %CI) of CKD was 1.23(1.02, 1.48) for every 1-unit increase in lnVAI' (Results)",
   "model_label": "Model 3",
   "covariates_in_this_model": [
    "age",
    "gender",
    "race",
    "education",
    "physical activity",
    "smoking",
    "CVD history",
    "cancer",
    "diabetes",
    "hypertension",
    "SBP",
    "DBP",
    "glycohemoglobin",
    "total cholesterol",
    "low-density lipoprotein cholesterol (LDL-C)",
    "cholesterol-lowering medications"
   ],
   "n_analytic": 6085,
   "n_quote": "Finally, 6085 were left for further analysis. (Data source); Table 2 caption: 'Association between VAI and CKD in 6085 US adults aged 60 years or older, NHANES 2011–2018.'",
   "events": null
  },
  "cycles": [
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=60",
   "inclusion": "NHANES 2011-2018 participants aged 60 years or older with renal function data (serum creatinine and urine ACR) and the measurements needed to calculate VAI",
   "exclusions": [
    "Start: 39,156 participants in 2011-2018; 7683 were aged 60 years or older (younger participants excluded)",
    "902 excluded for missing renal function data (serum creatinine and urine ACR), leaving 6781",
    "696 excluded with empty VAI, leaving 6085"
   ],
   "quote": "The inclusion criteria for the current study were participants aged 60 years or older and having available measurements for the calculation of VAI. Fig. 1 illustrates the selection process of study groups. We initially selected 39,156 participants, in which 7683 were aged 60 years or older. Next, we excluded 902 participants with missing data on renal function, including serum creatinine and urine albumin and creatinine ratio (ACR). Then, we excluded 696 participants with empty VAI. Finally, 6085 were left for further analysis. (Data source)"
  },
  "exposure": {
   "definition": "Visceral adiposity index (Amato et al., 2010), sex-specific: males {waist circumference/[39.68 + (1.88 x BMI)]} x (triglycerides/1.03) x (1.31/HDL-C); females {waist circumference/[36.58 + (1.89 x BMI)]} x (triglycerides/0.81) x (1.52/HDL-C). Waist in cm, BMI in kg/m2; triglycerides and HDL-C units are not stated in the formula (Amato's original uses mmol/L; Table 1 reports both in mmol/L). Headline uses lnVAI (natural log), continuous.",
   "nhanes_variables": [
    "BMXWAIST",
    "BMXBMI",
    "LBDSTRSI",
    "LBDTRSI",
    "LBDHDDSI",
    "RIAGENDR"
   ],
   "nhanes_files": [
    "BMX",
    "BIOPRO",
    "TRIGLY",
    "HDL",
    "DEMO"
   ],
   "transform": "ln (natural log), continuous, per 1 unit; quartiles of VAI for the categorical analysis",
   "categories": "VAI quartiles: 'Participants were divided into four groups according to their VAI quartiles: Q1 (<1.1), Q2 (1.1–1.8), Q3 (1.8–3.0) and Q4 (>3.0).' (Results, Participant characteristics)",
   "quote": "VAI was calculated based on the gender-specific mathematical model: males: {waist circumference/[39.68 + (1.88 × BMI)]}× (triglycerides/1.03) × (1.31/HDL-C); females: {waist circumference/[36.58 +(1.89 × BMI)]} × (triglycerides/0.81) × (1.52/HDL-C) (Amato et al., 2010). BMI and waist circumference were measured during the physical examination. Triglycerides and HDL-C values were available during the laboratory data section. (Definition of the VAI and CKD) [...] The distributions of the VAI were ln-transformed to reduce skewness. (Statistical analysis)"
  },
  "outcome": {
   "definition": "CKD = eGFR < 60 mL/min/1.73 m2 (CKD-EPI 2009 creatinine equation) and/or urinary albumin-to-creatinine ratio > 30 mg/g",
   "nhanes_variables": [
    "LBXSCR",
    "URXUMA",
    "URXUCR",
    "URDACT",
    "RIDAGEYR",
    "RIAGENDR",
    "RIDRETH3"
   ],
   "nhanes_files": [
    "BIOPRO",
    "ALB_CR",
    "DEMO"
   ],
   "quote": "CKD was defined as an eGFR <60 ml/min/1.73 m2 and/or a urinary ACR >30 mg/g (Andrassy, 2013). Serum creatinine and urine ACR were measured. Serum creatinine was used to calculate the estimated glomerular filtration rate (eGFR) with the CKD-EPI equation (Levey et al., 2009). (Definition of the VAI and CKD)"
  },
  "covariates": [
   {
    "name": "age",
    "coding": "years (Table 1 mean); model coding not stated",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "gender",
    "coding": "male/female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "race",
    "coding": "Table 1: Mexican American, Other Hispanic, Non-Hispanic white, Non-Hispanic black, Other races (other non-Hispanic race including non-Hispanic multiracial)",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "education",
    "coding": "Table 1: <high school, high school, >high school",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "physical activity",
    "coding": "GPAQ-based MET-min/week (vigorous 8 MET; moderate and walking or bicycling for transportation 4 MET); high >= 600 vs low < 600 MET-min/week",
    "nhanes_variables": [
     "PAQ605",
     "PAQ610",
     "PAD615",
     "PAQ620",
     "PAQ625",
     "PAD630",
     "PAQ635",
     "PAQ640",
     "PAD645",
     "PAQ650",
     "PAQ655",
     "PAD660",
     "PAQ665",
     "PAQ670",
     "PAD675"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": false,
    "note": "PAQ_L 2021-2023 has only leisure-time moderate and vigorous activity (PAD790Q/U, PAD800, PAD810Q/U, PAD820) and sedentary minutes; no work or transportation activity and no GPAQ, so this MET score cannot be rebuilt. Also, Table 2's note omits physical activity from Model 3 although the Methods, Results and figure captions include it."
   },
   {
    "name": "smoking",
    "coding": "Table 1: current smoker, ex-smoker, non-smoker",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "CVD history",
    "coding": "self-reported coronary heart disease, angina, stroke, or myocardial infarction (yes/no)",
    "nhanes_variables": [
     "MCQ160C",
     "MCQ160D",
     "MCQ160E",
     "MCQ160F"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "cancer",
    "coding": "self-reported cancer history (yes/no)",
    "nhanes_variables": [
     "MCQ220"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "diabetes",
    "coding": "HbA1c >= 6.5%, self-reported physician diagnosis, or current use of medications for diabetes or high blood sugar (yes/no)",
    "nhanes_variables": [
     "LBXGH",
     "DIQ010",
     "DIQ050",
     "DIQ070"
    ],
    "nhanes_files": [
     "GHB",
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "hypertension",
    "coding": "average SBP >= 140 and/or average DBP >= 90 mmHg, current antihypertensive medication, or self-reported physician diagnosis (yes/no)",
    "nhanes_variables": [
     "BPXSY1",
     "BPXSY2",
     "BPXSY3",
     "BPXDI1",
     "BPXDI2",
     "BPXDI3",
     "BPQ020",
     "BPQ050A",
     "BPXOSY1",
     "BPXOSY2",
     "BPXOSY3",
     "BPXODI1",
     "BPXODI2",
     "BPXODI3",
     "BPQ150"
    ],
    "nhanes_files": [
     "BPX",
     "BPXO",
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "2011-2018 used auscultatory BPX and BPQ050A; 2021-2023 has oscillometric BPXO_L and BPQ150 (same construct and units)."
   },
   {
    "name": "SBP",
    "coding": "mmHg, continuous (average of readings; number not stated)",
    "nhanes_variables": [
     "BPXSY1",
     "BPXSY2",
     "BPXSY3",
     "BPXOSY1",
     "BPXOSY2",
     "BPXOSY3"
    ],
    "nhanes_files": [
     "BPX",
     "BPXO"
    ],
    "in_2021_2023": true,
    "note": "Oscillometric in 2021-2023."
   },
   {
    "name": "DBP",
    "coding": "mmHg, continuous",
    "nhanes_variables": [
     "BPXDI1",
     "BPXDI2",
     "BPXDI3",
     "BPXODI1",
     "BPXODI2",
     "BPXODI3"
    ],
    "nhanes_files": [
     "BPX",
     "BPXO"
    ],
    "in_2021_2023": true,
    "note": "Oscillometric in 2021-2023."
   },
   {
    "name": "glycohemoglobin",
    "coding": "%, continuous",
    "nhanes_variables": [
     "LBXGH"
    ],
    "nhanes_files": [
     "GHB"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "total cholesterol",
    "coding": "mmol/L, continuous (Table 1)",
    "nhanes_variables": [
     "LBDTCSI"
    ],
    "nhanes_files": [
     "TCHOL"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "LDL-C",
    "coding": "mmol/L, continuous (Table 1); calculation not stated",
    "nhanes_variables": [
     "LBDLDLSI"
    ],
    "nhanes_files": [
     "TRIGLY"
    ],
    "in_2021_2023": true,
    "note": "NHANES LDL-C exists only for the fasting subsample; how non-fasting participants got an LDL-C value is not stated. TRIGLY_L also has Martin-Hopkins and NIH equation values."
   },
   {
    "name": "cholesterol-lowering medications",
    "coding": "yes/no (self-report)",
    "nhanes_variables": [
     "BPQ100D",
     "BPQ101D"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "BPQ100D in 2011-2018; BPQ101D in 2021-2023."
   }
  ],
  "design": {
   "weights": "Sampling weights are said to be used for the nationally representative descriptive estimates; the weight variable (e.g. WTMEC2YR/4) is not named, and whether the regression models were weighted is not stated",
   "strata_psu": "not stated",
   "quote": "And sampling weights were used to produce nationally representative prevalence estimates for the non-institutionalized US population according to the NHANES introduction. Continuous variables were shown as survey-weighted mean (SE). Categorical variables were shown as survey-weighted percentage and missing values were categorized as a group. (Statistical analysis)",
   "software": "R 4.2.0",
   "missing_data": "missing category for categorical covariates; participants missing renal function or VAI excluded; handling of missing continuous covariates unstated",
   "quote_missing": "Categorical variables were shown as survey-weighted percentage and missing values were categorized as a group."
  },
  "model": {
   "family": "logistic",
   "weighted": null,
   "quote": "Logistic regression models were used to analyse the association between VAI and CKD. VAI was regarded as a categorical variable (divided into quarters). [...] In the logistic regression models, higher lnVAI was associated with a higher risk of CKD (Table 2)."
  },
  "unstated": [
   "whether the logistic regression models (and GAM, threshold and subgroup models) were survey-weighted; which weight (e.g. WTMEC2YR divided by 4) and whether strata and PSUs were used",
   "the triglyceride source for VAI (fasting-subsample TRIGLY or non-fasting standard biochemistry BIOPRO) and the units used in the VAI formula",
   "how LDL-C was obtained for participants outside the fasting subsample",
   "CKD-EPI 2009 details (race coefficient applied; which creatinine variable)",
   "whether the ACR cutoff was > 30 or >= 30 mg/g",
   "how missing continuous covariates were handled (only categorical missing-as-a-group is stated)",
   "coding of each Model 3 covariate (age, SBP, DBP, glycohemoglobin, cholesterol, LDL-C presumably continuous; race, education, smoking categorical as in Table 1; physical activity binary at 600 MET-min/week)",
   "whether physical activity is in Model 3 (Methods, Results and figure captions say yes; Table 2 and Table 3 notes omit it)",
   "number of blood pressure readings averaged",
   "which questionnaire items defined antihypertensive, diabetes and cholesterol-lowering medication use",
   "whether VAI quartile cutpoints were weighted (group sizes 1521, 1521, 1521, 1522 suggest unweighted sample quartiles)"
  ],
  "notes": "Model 3 covariates are listed as in the Methods and in the Results sentence that reports the headline (both include physical activity: 'after adjusting for covariates including age, gender, race, education, physical activity, smoking, comorbidities, SBP, DBP, glycohemoglobin, total cholesterol, LDL-C and cholesterol-lowering medications'); the Table 2 and Table 3 notes omit physical activity. Table 2, Model 2, row Q3 prints '1.04(0.81, 1.33)' with P 0.017, the same estimate as Q2 in Model 2 (P 0.759): likely a copy error. The Results say the CKD prevalence rose 'among participants with a lnVAI between −0.6 and 1.4' while the abstract and Table 3 use 1.6. 'BMC' in the Results means BMI. The covariate-selection sentence ('changed the estimates of activities on mortality by more than 10%') is template text. The starting count of 39,156 equals the NHANES documentation counts for DEMO_G, DEMO_H, DEMO_I and DEMO_J (9,756 + 10,175 + 9,971 + 9,254), so no file overlap. Inference, not stated: only 696 of 6781 were lost for missing VAI, so triglycerides probably came from the non-fasting standard biochemistry profile (the fasting subsample is about half of examinees). Subgroup interactions were tested by Wald and likelihood ratio tests and the inflection points found by a recursive likelihood search, which are usually unweighted procedures. Supplement mmc1.docx (fetched from the publisher to scratchpad/dl/supp/row121/; not needed for the headline): Table S1 has BMI and waist quartiles only; Tables S2 and S3 use the narrower outcome eGFR < 60 (renal insufficiency), e.g. Table S3 '| lnVAI | 1.21 (0.97, 150) | .096 |' (upper bound printed '150').",
  "adjudication": null
 },
 {
  "id": "row311",
  "rank": 43,
  "row": 311,
  "doi": "10.1097/MD.0000000000039258",
  "pmcid": "PMC11315559",
  "title": "Association between triglyceride-glucose index and depression in patients with type 2 diabetes: A cross-sectional study from NHANES",
  "authors": [
   "Ren, Jiaju",
   "Lv, Cheng",
   "Wang, Jia"
  ],
  "year": 2024,
  "journal": "Medicine",
  "table_a": {
   "predictor": "Triglyceride glucose index",
   "condition": "Depression",
   "population": "US adults with type 2 diabetes"
  },
  "headline": {
   "abstract_quote": "After adjusting for age, gender, BMI, smoking, alcohol consumption, congestive heart failure, and coronary heart disease, a significant positive association was found between the TyG index and the prevalence of depression in individuals with type 2 diabetes (OR = 1.54, 95% CI: 1.21–1.95).",
   "table_location": "Table 3, column 'Model III' (95% CI and P value), row 'TyG'",
   "table_quote": "| TyG | 1.60 (1.29, 1.99) | <.001 | 1.64 (1.30, 2.07) | <.001 | 1.54 (1.21, 1.95) | <.001 |",
   "measure": "OR",
   "estimate": 1.54,
   "ci_low": 1.21,
   "ci_high": 1.95,
   "p_value": "<.001",
   "exposure_contrast": "per 1-unit increase in the TyG index (continuous): 'For every unit increase in the TyG index, the risk of developing depression increased by 54% (OR = 1.54, 95% CI: 1.21–1.95).' (Results 3.2)",
   "model_label": "Model III",
   "covariates_in_this_model": [
    "age",
    "sex",
    "marriage",
    "race",
    "body mass index (BMI)",
    "smoking",
    "alcohol",
    "congestive heart failure (CHF)",
    "coronary heart disease (CAD)"
   ],
   "n_analytic": 3225,
   "n_quote": "This process resulted in a final sample size of 3225 participants (Fig. 1). (2.1 Study design and population)",
   "events": 364
  },
  "cycles": [
   "2005-2006",
   "2007-2008",
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018",
   "2019-2020 (as stated; the Figure 1 starting count matches the 2017-March 2020 pre-pandemic files, see notes)"
  ],
  "population": {
   "age": ">=18",
   "inclusion": "NHANES 2005-2020 participants aged 18 or older with type 2 diabetes and complete TyG, PHQ-9 and covariate data",
   "exclusions": [
    "Start: 'Participants in NHANES from 2005-2020 N=85750' (Figure 1)",
    "'Missing data or imcomplete data on TyG index N=59891', leaving 'Participants with complete TyG index N=25859' (Figure 1)",
    "'Missing data or imcomplete data on PHQ9 N=5475', leaving 'Participants with complete PHQ9 data N=20384' (Figure 1)",
    "'Excluding participants with non-diabetes and missing covariates N=17159', leaving 'Final participants N=3225' (Figure 1)",
    "Age < 18 excluded (text); no separate count given"
   ],
   "quote": "Our study population consisted of individuals aged 18 years and older. To ensure data quality, type 2 diabetic participants with missing values for TyG, Patient Health Questionnaire-9 (PHQ-9), and other study variables were excluded from the analyses. This process resulted in a final sample size of 3225 participants (Fig. 1). (2.1). Figure 1 box labels were read from the figure image (fetched from Europe PMC, saved under scratchpad/dl/supp/row311/unz/); 'imcomplete' is as printed."
  },
  "exposure": {
   "definition": "TyG index = ln[fasting triglycerides (mg/dL) x fasting glucose (mg/dL)/2], natural log; continuous per 1 unit for the headline; quartiles for the categorical analysis",
   "nhanes_variables": [
    "LBXTR",
    "LBXGLU",
    "LBXTLG"
   ],
   "nhanes_files": [
    "TRIGLY",
    "GLU"
   ],
   "transform": "none beyond the ln in the index definition; per 1 unit of TyG",
   "categories": "Quartiles: 'the TyG index was categorized into 4 groups based on quartiles: Q1 < 8.61, Q2 (8.61,9.03), Q3 (9.03,9.47), and Q4 (>9.47).' (2.2)",
   "quote": "Plasma glucose and serum triglyceride levels were measured using an automated analyzer. The TyG index, which quantifies the relationship between fasting triglycerides (mg/dL) and fasting glucose (mg/dL), was calculated using the formula: [...] TyG = Ln[fasting triglycerides (mg/dL) × fasting glucose (mg/dL)/2]. (2.2 Primary research variables)"
  },
  "outcome": {
   "definition": "Depression = PHQ-9 total score >= 10 (nondepressed below 10)",
   "nhanes_variables": [
    "DPQ010",
    "DPQ020",
    "DPQ030",
    "DPQ040",
    "DPQ050",
    "DPQ060",
    "DPQ070",
    "DPQ080",
    "DPQ090"
   ],
   "nhanes_files": [
    "DPQ"
   ],
   "quote": "Participants’ depressive status was assessed using the PHQ-9. A PHQ-9 score of ≥10 indicated the presence of depression, and individuals meeting this criterion were categorized as depressed. Conversely, those below the threshold were considered nondepressed. (2.2)"
  },
  "covariates": [
   {
    "name": "age",
    "coding": "years (Table 1 mean); model coding not stated",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "sex",
    "coding": "male/female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "marriage",
    "coding": "Table 1: Married, Never married, Divorced",
    "nhanes_variables": [
     "DMDMARTL",
     "DMDMARTZ"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "2005-2016 DMDMARTL has six levels; how widowed, separated and living with partner were mapped is not stated. DMDMARTZ (2017-2020 pre-pandemic and 2021-2023) has three levels: married/living with partner, widowed/divorced/separated, never married. Asked of ages 20+."
   },
   {
    "name": "race",
    "coding": "Table 1: Non-Hispanic Black, Non-Hispanic White, Mexican American, Other Hispanic, Other Race",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "body mass index (BMI)",
    "coding": "normal (<25.0), overweight (25.0-29.9), obese (>30.0), as printed",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Categories leave BMI exactly 30.0 unassigned as printed; whether BMI entered Model III as categories is not stated."
   },
   {
    "name": "smoking",
    "coding": "nonsmokers, former smokers, current smokers",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "alcohol",
    "coding": "none, moderate, heavy by average weekly drinks in the past 12 months (0, 1-8, >= 8)",
    "nhanes_variables": [
     "ALQ120Q",
     "ALQ120U",
     "ALQ121",
     "ALQ130"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "ALQ120Q/ALQ120U before 2017-2018, ALQ121 from 2017-2018 on; ALQ121 and ALQ130 are in ALQ_L. The 1-8 and >= 8 bounds overlap at 8 as printed."
   },
   {
    "name": "congestive heart failure (CHF)",
    "coding": "self-reported yes/no",
    "nhanes_variables": [
     "MCQ160B"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "coronary heart disease (CAD)",
    "coding": "self-reported yes/no",
    "nhanes_variables": [
     "MCQ160C"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   }
  ],
  "design": {
   "weights": "'recommended weights of the database'; the weight variable (presumably the fasting-subsample weight) and how 8 cycles were combined are not stated",
   "strata_psu": "not stated",
   "quote": "The analyses was conducted using the recommended weights of the database, and R (version 4.2.1) was used for all data processing and analyses. (2.4)",
   "software": "R 4.2.1",
   "missing_data": "complete case",
   "quote_missing": "To ensure data quality, type 2 diabetic participants with missing values for TyG, Patient Health Questionnaire-9 (PHQ-9), and other study variables were excluded from the analyses."
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "Table 3 presents the results of the multivariable logistic regression analyses. [...] The analyses was conducted using the recommended weights of the database"
  },
  "unstated": [
   "which weight was used (presumably the fasting-subsample weight WTSAF2YR), how it was rescaled across 8 cycles, and whether strata and PSUs were used",
   "which files make up the eighth cycle ('2019-2020'; the public release is the 2017-March 2020 pre-pandemic file, which repeats the 2017-2018 participants) and whether duplicate participants were removed",
   "the HbA1c threshold actually applied (printed as '>= 5.29%')",
   "how type 2 diabetes was separated from type 1",
   "minimum fasting time for the triglyceride and glucose values",
   "how PHQ-9 item nonresponse was handled",
   "how each Model III covariate was coded (age continuous or grouped; BMI categories or continuous; mapping of marital categories; smoking and alcohol definitions from NHANES items)",
   "how quartile cutpoints were computed (weighted or unweighted)",
   "which questionnaire items defined hypoglycemic therapy, CHF and CAD"
  ],
  "notes": "Overlapping files, likely: Figure 1 starts from 'Participants in NHANES from 2005-2020 N=85750', which equals the NHANES documentation counts for DEMO_D to DEMO_J (10,348 + 10,149 + 10,537 + 9,756 + 10,175 + 9,971 + 9,254 = 70,190) plus P_DEMO (15,560). So the eighth 'cycle' is the 2017-March 2020 pre-pandemic release, which contains the 2017-2018 participants again (the same overlap that got the row 102 paper retracted); whether duplicates stayed in the final 3225 is not stated. The printed HbA1c threshold 'hemoglobin A1c ≥ 5.29%' is not a diabetes cutoff; since 3225 of 20,384 remained after excluding non-diabetes and missing covariates, the usual 6.5% was probably applied (inference). Table 1 labels TG, HDL and LDL as mmol/L but the values ('| TG (mmol/L) | 142.66 (2.08) | 70.10 (0.93) | 111.41 (1.27) | 152.51 (1.80) | 228.39 (2.99) | <.001 |') are mg/dL. Table 1 smoking gives Now 1664 (51.28%) and Never 533 (15.93%), implausible for US adults with diabetes, so the labels may be swapped (inference). Table 1 alcohol counts sum to 2734 (1204 + 1162 + 368), not 3225; Table 2's sum to 3225 (1204 + 1530 + 491). The abstract's covariate list omits race and marital status, which Model III includes. Section 2.3 lists glucose, triglyceride, HDL and LDL among confounders and says 'All these potential confounders were included in the analyses', but no model definition includes them. Subgroup interactions used likelihood ratio tests. The ethics note cites approval by the Ethics Committee of Emergency General Hospital (K19-9).",
  "adjudication": null
 },
 {
  "id": "row040",
  "rank": 47,
  "row": 40,
  "doi": "10.1186/s12944-024-02262-2",
  "pmcid": "PMC11340038",
  "title": "Association of non-high-density lipoprotein cholesterol to high-density lipoprotein cholesterol ratio (NHHR) and gallstones among US adults aged ≤ 50 years: a cross-sectional study from NHANES 2017–2020",
  "authors": [
   "Cheng, Quankai",
   "Wang, Ziming",
   "Zhong, Haicheng",
   "Zhou, Sheng",
   "Liu, Chang",
   "Sun, Jingjing",
   "Zhao, Sihai",
   "Deng, Jie"
  ],
  "year": 2024,
  "journal": "Lipids in Health and Disease",
  "table_a": {
   "predictor": "Non-high-density lipoprotein cholesterol to high-density lipoprotein cholesterol ratio (NHHR)",
   "condition": "Gallstones",
   "population": "US middle-aged and older adults"
  },
  "headline": {
   "abstract_quote": "In multivariate logistic regression that accounted for all factors, there was a 77% increase in the likelihood of gallstones for every unit rise in lnNHHR (OR 1.77 [CI 1.11–2.83]).",
   "table_location": "Table 2 (Association between NHHR and the odds of gallstones), Model 3 columns 'OR (95% CI)' and 'P-value', row 'Ln-NHHR'",
   "table_quote": "| Ln-NHHR <colspan=2> | 1.42 (1.00,2.00) | 0.0487 | 1.80 (1.22,2.66) | 0.0032 | 1.77 (1.11,2.83) | 0.0163 |",
   "measure": "OR",
   "estimate": 1.77,
   "ci_low": 1.11,
   "ci_high": 2.83,
   "p_value": "0.0163",
   "exposure_contrast": "per 1-unit increase in natural-log-transformed NHHR (lnNHHR), NHHR as a continuous variable",
   "model_label": "Model 3 (fully adjusted)",
   "covariates_in_this_model": [
    "gender",
    "age",
    "race",
    "PIR",
    "education level",
    "BMI",
    "smoking status",
    "alcohol consumption",
    "total cholesterol",
    "HDL-C",
    "sedentary time",
    "hypertension",
    "diabetes",
    "CHD",
    "heart attack",
    "COPD",
    "physical activity"
   ],
   "n_analytic": 3772,
   "n_quote": "Materials and methods, Research design: \"In the end, 3,772 people were part of the research.\"; Abstract, Methods: \"3,772 people, all under 50, were included in this study\"; Fig. 1 (figure image): \"3,772 participants included in the final analysis\".",
   "events": null
  },
  "cycles": [
   "2017-March 2020"
  ],
  "population": {
   "age": "<= 50 (participants older than 50 excluded); no lower bound stated, but the gallstone item MCQ550 is asked only from age 20, so effectively 20-50",
   "inclusion": "NHANES 2017-March 2020 participants aged 50 or younger with total cholesterol, HDL-C and gallstone questionnaire data",
   "exclusions": [
    "Older than 50 years of age (N=4841), leaving 10,719 (Fig. 1)",
    "Without total cholesterol and high-density lipoprotein cholesterol data (N=4036), leaving 6,683 (Fig. 1)",
    "Without gallstone data (N=2911), leaving 3,772 (Fig. 1)"
   ],
   "quote": "Research design: \"Initially, this phase of the study included 15,560 participants. However, we excluded middle-aged and older participants over 50 years of age. Furthermore, 2,911 individuals and 4,036 participants were eliminated for lacking gallstone data and cholesterol data, respectively. In the end, 3,772 people were part of the research.\" Fig. 1 (flowchart image): \"15,560 participants in 2017-2020 NHANES\"; \"Exclusion of participants older than 50 years of age (N=4841)\"; \"10,719 participants\"; \"Exclusion of participants without total cholesterol and high-density lipoprotein cholesterol data (N=4036)\"; \"6,683 participants\"; \"Exclusion of participants Without gallstone data (N=2911)\"; \"3,772 participants included in the final analysis\"."
  },
  "exposure": {
   "definition": "NHHR = non-HDL-C / HDL-C, where non-HDL-C = total cholesterol minus HDL-C (unit-free ratio). The headline uses its natural log (lnNHHR) as a continuous variable.",
   "nhanes_variables": [
    "LBXTC",
    "LBDHDD"
   ],
   "nhanes_files": [
    "TCHOL",
    "HDL"
   ],
   "transform": "natural log (ln), per 1 unit",
   "categories": "Quartiles (secondary analysis, not the headline), Table 1 header: \"| Q1(≤ 1.79) | Q2(1.79–2.51) | Q3(2.51–3.46) | Q4(≥ 3.46) |\"",
   "quote": "Definition of NHHR and gallstones: \"The definition of NHHR was given as the ratio of non-high-density lipoprotein cholesterol (NHDL-C) to high-density lipoprotein cholesterol (HDL-C), where NHDL-C was determined by subtracting the HDL-C level from the overall cholesterol amount [25].\" Statistical analysis: \"the NHHR was ln-transformed to ensure its normal distribution\"."
  },
  "outcome": {
   "definition": "Self-reported doctor-diagnosed gallstones, yes vs no",
   "nhanes_variables": [
    "MCQ550"
   ],
   "nhanes_files": [
    "MCQ"
   ],
   "quote": "Definition of NHHR and gallstones: \"Based on the query, gallstone statistics were taken from the MCQ questionnaire, “Has a doctor or other health professional ever told you that you have gallstones?” to ascertain whether gallstones are present [26].\""
  },
  "covariates": [
   {
    "name": "gender",
    "coding": "Male, Female (Table 1)",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "age",
    "coding": "years, continuous (Table 1 mean)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "race",
    "coding": "Mexican American, Other Hispanic, Non-Hispanic White, Non-Hispanic Black, Other Race (Table 1)",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "How race entered Model 3 is not stated (dummy variables or one 1-5 code; the Fig. 7 nomogram uses race coded 1-5 as a single variable)."
   },
   {
    "name": "PIR",
    "coding": "≤ 1.3, 1.3–3.5, ≥ 3.5 (Table 1)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Whether entered as these categories or continuous is not stated."
   },
   {
    "name": "education level",
    "coding": "Less than high school; High school or equivalent; College graduate or above (Table 1)",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Table 1's three levels sum to 100%, so DMDEDUC2's 'some college or AA degree' must be folded into one of them; the mapping is not stated."
   },
   {
    "name": "BMI",
    "coding": "Fig. 2: normal, ≤25 kg/m2; overweight, 25-30 kg/m2; obese, ≥30 kg/m2",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Whether Model 3 used the categories or continuous BMI is not stated (Table 3's prediction model reports a per-unit OR)."
   },
   {
    "name": "smoking status",
    "coding": "Fig. 2: Yes: smoked at least 100 cigarettes in life",
    "nhanes_variables": [
     "SMQ020"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "alcohol consumption",
    "coding": "Fig. 2: Yes: ever have 4/5 or more drinks every day",
    "nhanes_variables": [
     "ALQ151"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "ALQ151 ('Was there ever a time or times in your life when you drank 4/5 or more drinks of any kind of alcoholic beverage almost every day?') is in both the 2017-March 2020 file and ALQ_L."
   },
   {
    "name": "total cholesterol",
    "coding": "mmol/L, continuous",
    "nhanes_variables": [
     "LBDTCSI"
    ],
    "nhanes_files": [
     "TCHOL"
    ],
    "in_2021_2023": true,
    "note": "Component of the exposure ratio."
   },
   {
    "name": "HDL-C",
    "coding": "mmol/L, continuous",
    "nhanes_variables": [
     "LBDHDDSI"
    ],
    "nhanes_files": [
     "HDL"
    ],
    "in_2021_2023": true,
    "note": "Component of the exposure ratio."
   },
   {
    "name": "sedentary time",
    "coding": "minutes, continuous",
    "nhanes_variables": [
     "PAD680"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": true,
    "note": "PAD680 (minutes sedentary activity) is in both the 2017-March 2020 file and PAQ_L."
   },
   {
    "name": "hypertension",
    "coding": "Yes/No; definition not given (Fig. 2 omits it)",
    "nhanes_variables": [
     "BPQ020"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Definition unstated; self-report BPQ020 and oscillometric blood pressure (BPXO_L, the same device type as 2017-March 2020) both exist in 2021-2023."
   },
   {
    "name": "diabetes",
    "coding": "Yes/No; definition not given",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "Definition unstated (self-report DIQ010 presumed; glucose or HbA1c criteria would also be constructible)."
   },
   {
    "name": "CHD",
    "coding": "Yes/No",
    "nhanes_variables": [
     "MCQ160c"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "Source item not stated; MCQ160c (coronary heart disease) presumed."
   },
   {
    "name": "heart attack",
    "coding": "Yes/No",
    "nhanes_variables": [
     "MCQ160e"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "Source item not stated; MCQ160e presumed."
   },
   {
    "name": "COPD",
    "coding": "Yes/No",
    "nhanes_variables": [
     "MCQ160p"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "The 2017-March 2020 MCQ file's only COPD item is MCQ160p ('COPD, emphysema, or chronic bronchitis'), the same item as MCQ160p in MCQ_L."
   },
   {
    "name": "physical activity",
    "coding": "Fig. 2: Yes: work involves moderate-intensity activities that result in small increases in respiration or heart rate",
    "nhanes_variables": [
     "PAQ620"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": false,
    "note": "This is the work item PAQ620 ('Does your work involve moderate-intensity activity that causes small increases in breathing or heart rate...'); PAQ_L in 2021-2023 has only leisure-time moderate and vigorous activity and sedentary minutes, no work activity."
   }
  ],
  "design": {
   "weights": "Not stated which weight, or whether the logistic models were weighted. Methods mention only that NHANES uses sample weights and that Table 1 used weighted chi-square and t-tests. The pre-pandemic MEC weight would be WTMECPRP.",
   "strata_psu": "not stated",
   "quote": "Statistical analysis: \"Considering that the NHANES program employs a sample-weighted statistical method.\" ... \"We then used a weighted chi-square test for categorical variables and a weighted Student’s t-test for continuous variables.\"",
   "software": "R 4.3.1. Statistical analysis: \"The R program (4.3.1) was utilized for our study\"",
   "missing_data": "imputation (multiple imputation by chained equations, 5 imputations, for missing covariates; participants missing NHHR components or gallstone data were excluded)",
   "quote_missing": "Statistical analysis: \"Missing covariates were addressed by performing multiple imputations with five replications, a chained equation method was used to handle missing data across these replications [28].\""
  },
  "model": {
   "family": "logistic",
   "weighted": null,
   "quote": "Statistical analysis: \"we created 3 models (one unadjusted for covariates, one adjusted for population baseline data, and one adjusted for all covariates), explored their correlation using multivariate logistic regression, and established a linear correlation between NHHR and gallstones using smoothed curve fitting.\""
  },
  "unstated": [
   "Whether the logistic models were survey-weighted, which weight (the 2017-March 2020 pre-pandemic MEC weight is WTMECPRP), and whether strata (SDMVSTRA) and PSUs (SDMVPSU) were used",
   "The lower age bound (MCQ550 is asked only from age 20, so effectively 20-50)",
   "Definition of hypertension (self-report BPQ020, measured blood pressure, medication use, or a combination)",
   "Definition of diabetes (self-report DIQ010 only, or glucose/HbA1c/medication criteria too; handling of 'borderline')",
   "Source items for CHD, heart attack and COPD (presumably MCQ160c, MCQ160e, MCQ160p)",
   "How race, PIR, education and BMI entered Model 3 (categorical with which reference levels, or continuous)",
   "How 'Refused' and 'Don't know' answers to MCQ550 and to covariate items were handled",
   "Which variables entered the multiple-imputation model and how estimates were pooled across the five imputations",
   "Whether people who do not work (PAQ620 'No') are coded as physically inactive",
   "Whether pregnant women were excluded (no such exclusion is reported)",
   "Whether ALQ151 'No' includes never-drinkers who skipped the item",
   "The number of gallstone cases among the 3,772 analyzed",
   "Why Table 1 totals 2,117 participants rather than 3,772 (see notes)"
  ],
  "notes": "Table A's population label ('US middle-aged and older adults') contradicts the paper, which excluded everyone older than 50 (title: 'US adults aged ≤ 50 years'). Fig. 1 (flowchart) and Fig. 2 (covariate definitions) are images missing from the text conversion; they were read from the PMC figure images saved under scratchpad/dl/supp/row040/ (Fig1.jpg, Fig2.jpg). The 2017-March 2020 documentation pages P_MCQ.htm, P_ALQ.htm and P_PAQ.htm saved there confirm that MCQ160p, ALQ151 and PAQ620 are the items named. Analytic N doubt: Table 2's Model 1 quartile ORs and CIs equal the unweighted crude odds ratios computed from Table 1's counts (n = 2,117): computed here Q2 1.134 (0.693, 1.855), Q3 1.339 (0.832, 2.157), Q4 1.511 (0.948, 2.409), printed 1.13 (0.69,1.86), 1.34 (0.83,2.16), 1.51 (0.95,2.41). This suggests the regressions were unweighted and fitted on about 2,117 participants rather than 3,772; the Ln-NHHR row cannot be checked this way. Other Table 1 inconsistencies: the text says 48.13% male but the Gender rows give 1,142 of 2,117 (53.9%); the PIR rows' column percentages exceed 100% in Q2 (26.79 + 44.42 + 47.64) and Q3 (46.69 + 47.17 + 27.79); continuous variables are called mean ± standard error but are SD-sized. Model 3 adjusts for total cholesterol and HDL-C, the two components of the exposure ratio. ALT is listed among covariates in Methods but is not in Model 3 (Table 2 note c). Methods say lipids were measured on a 'Roche Cobas 6000 (c501 module)' and that blood was drawn 'in a fasting state', but the flowchart does not restrict to the fasting subsample (6,683 had total cholesterol and HDL-C), so the full phlebotomy sample appears to be used. The abstract says 'all under 50', the title says '≤ 50', and Fig. 1 excludes those 'older than 50', so age 50 appears to be included. The 15,560 starting count matches the 2017-March 2020 pre-pandemic demographics file, so the combined pre-pandemic (P_) files were used. The prediction-model part (LASSO, Table 3, nomogram, AUC 0.785) is separate from the headline.",
  "adjudication": null
 },
 {
  "id": "row257",
  "rank": 48,
  "row": 257,
  "doi": "10.1016/j.waojou.2024.100900",
  "pmcid": "PMC11053303",
  "title": "Association between dietary zinc intake and asthma in overweight or obese children and adolescents: A cross-sectional analysis of NHANES",
  "authors": [
   "Cheng, Chuhan",
   "Lin, Jing",
   "Zhang, Zihan",
   "Zhang, Liyan"
  ],
  "year": 2024,
  "journal": "World Allergy Organization Journal",
  "table_a": {
   "predictor": "Dietary zinc intake",
   "condition": "Asthma",
   "population": "US obese children and adolescents"
  },
  "headline": {
   "abstract_quote": "After adjusting for all covariates in the multivariate logistic regression, compared with the lowest zinc intake group Q1(≤5.68 mg/day), the adjusted OR values for zinc intake and asthma in Q2 (5.69–8.36 mg/day), Q3 (8.37–11.95 mg/day), and Q4 (≥11.96 mg/day) were 0.78 (95% CI: 0.62–0.98, p = 0.03), 0.76 (95% CI: 0.6∼0.98, p = 0.032), 0.71 (95% CI: 0.53∼0.95, p = 0.022), respectively.",
   "table_location": "Table 3, Model 3, row Q4",
   "table_quote": "| Qc2(5.69∼8.36) | 1156 | 916 (79.2) | 0.85 (0.69∼1.05) | 0.126 | 0.84 (0.68∼1.03) | 0.097 | 0.79 (0.64∼0.99) | 0.037 | 0.78 (0.62∼0.98) | 0.03 |",
   "measure": "OR",
   "estimate": 0.71,
   "ci_low": 0.53,
   "ci_high": 0.95,
   "p_value": "0.022",
   "exposure_contrast": "quartile 4 (>= 11.96 mg/day) vs quartile 1 (<= 5.68 mg/day) of dietary zinc intake",
   "model_label": "Model 3",
   "covariates_in_this_model": [
    "age",
    "sex",
    "race and ethnicity",
    "PIR",
    "family asthma",
    "second-hand smoking",
    "EOPC",
    "WBC",
    "HGB",
    "calorie consumption",
    "protein consumption",
    "carbohydrate consumption",
    "sugar consumption",
    "fiber consumption",
    "fat consumption"
   ],
   "n_analytic": 4597,
   "n_quote": "Results, Study population: \"In all, the present study included 4597 individuals in total, of which 963 reported having asthma (Fig. 1).\" Abstract: \"A total of 4597 pediatrics and adolescents were enrolled, with 20.9% (963/4597) suffering from asthma.\"",
   "events": 963
  },
  "cycles": [
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-March 2020"
  ],
  "population": {
   "age": "<20 (lower bound not stated; the BMI category BMDBMIC exists only for ages 2-19, so effectively 2-19)",
   "inclusion": "Participants under 20 with complete asthma questionnaire, dietary zinc and BMI category data who were overweight (BMI 85th to 95th percentile) or obese (95th percentile and above)",
   "exclusions": [
    "Aged 20 or older: 26,280 of 45,462 respondents, leaving 19,182 under 20",
    "Incomplete asthma questionnaires (n = 1793)",
    "Incomplete dietary zinc intake information (n = 3396)",
    "No BMI information (n = 1112)",
    "Underweight or normal weight (n = 8284), leaving 4597 (963 with asthma)"
   ],
   "quote": "Inclusion and exclusion criteria: \"Participants under the age of 20 who completed the survey were incorporated. Individuals who had incomplete data regarding questionnaire responses for asthma, dietary zinc intake, and Body mass index (BMI) Category were eliminated from the study, as well as those who were either underweight or of normal weight.\" Results, Study population: \"Out of the 45,462 respondents who finished the poll, 26,280 were 20 years of age or older. Among the remaining 19,182 participants under 20, those with incomplete asthma questionnaires (n = 1793) and those who did not provide complete dietary zinc intake information (n = 3396) were not included. The study eliminated those who did not have BMI information (n = 1112) and those who were underweight or of normal weight (n = 8284). In all, the present study included 4597 individuals in total, of which 963 reported having asthma (Fig. 1).\" Overweight or obese: \"The data on BMI Category were derived from the examination data and there were 4 categories:21 underweight (BMI<5th percentile), normal weight (BMI 5th to 85th percentile), overweight (BMI 85th to 95th percentile), and obese (BMI⩾95th percentile).\""
  },
  "exposure": {
   "definition": "Total dietary zinc intake (mg/day) from foods and beverages in the 24-hour dietary recall (AMPM), in quartiles with Q1 as reference. Whether day 1 alone or a two-day mean was used is not stated; supplement zinc is not mentioned.",
   "nhanes_variables": [
    "DR1TZINC"
   ],
   "nhanes_files": [
    "DR1TOT"
   ],
   "transform": "categories: quartiles at fixed cutpoints, Q1 reference",
   "categories": "Q1 ≤5.68, Q2 5.69-8.36, Q3 8.37-11.95, Q4 ≥11.96 mg/day. Abstract: \"compared with the lowest zinc intake group Q1(≤5.68 mg/day), the adjusted OR values for zinc intake and asthma in Q2 (5.69–8.36 mg/day), Q3 (8.37–11.95 mg/day), and Q4 (≥11.96 mg/day)\"; Table 1 header: \"| Total (n = 4597) | Qa1(≤5.68) (n = 1141) | Qa2(5.69–8.36) (n = 1156) | Qa3(8.37–11.95) (n = 1147) | Qa4(≥11.96) (n = 1153) | p-Value |\"",
   "quote": "Dietary zinc intake: \"Dietary survey participants in NHANES were asked to report their consumption of food and beverages within a 24-h period. Data on dietary intake were collected between 2011 and 2020 using the Automated Multiple Pass Method (AMPM).\" ... \"The subjects were divided into 4 groups based on their dietary zinc consumption.\" Statistical analysis: \"The variable of interest was the dietary intake of zinc, and all analyses were performed according to quartiles of zinc intake.\""
  },
  "outcome": {
   "definition": "Ever told by a doctor or other health professional that they have asthma: yes = asthma, no = no asthma",
   "nhanes_variables": [
    "MCQ010"
   ],
   "nhanes_files": [
    "MCQ"
   ],
   "quote": "Asthma assessment: \"To identify participants with asthma, we assessed their answers to the question, “Has a doctor or other health professional ever told you that you have asthma?” in the medical condition questionnaire. Those who replied “yes” were classified as asthmatic, whereas those who responded “no” were classified as non-asthmatic.\""
  },
  "covariates": [
   {
    "name": "age",
    "coding": "years, continuous (Table 1 median (IQR); Table 2 per-year OR)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "sex",
    "coding": "Male (reference), Female (Table 2)",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "race and ethnicity",
    "coding": "Mexican American (reference), Other Hispanic, Non-Hispanic White, Non-Hispanic Black, Other/Multi-Racial (Table 2)",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "PIR",
    "coding": "low (PIR ⩽ 1.3, reference), medium (1.3–3.5), high (PIR >3.5) (Methods; Table 2 reference 'Low(<1.3)')",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Methods put 1.3 in 'low' (⩽ 1.3); Table 1 and 2 labels say 'Low(<1.3)'."
   },
   {
    "name": "family asthma",
    "coding": "No (reference), Yes (Table 2); definition not stated",
    "nhanes_variables": [
     "MCQ300b"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": false,
    "note": "Presumably the close-relative item MCQ300b ('were any of {SP's/your} close biological ... relatives ... ever told by a health professional that they had asthma?'), asked of children in the paper's cycles; MCQ_L in 2021-2023 has no family-history items."
   },
   {
    "name": "second-hand smoking",
    "coding": "Yes (reference), No (Table 2); definition not stated",
    "nhanes_variables": [
     "SMD460"
    ],
    "nhanes_files": [
     "SMQFAM"
    ],
    "in_2021_2023": true,
    "note": "Source not stated. Household smokers (SMD460, SMQFAM_L) and serum cotinine (LBXCOT, COT_L) both exist in 2021-2023; SMD460 is listed as the likely source, not a confirmed one."
   },
   {
    "name": "EOPC",
    "coding": "eosinophils percent, continuous",
    "nhanes_variables": [
     "LBXEOPCT"
    ],
    "nhanes_files": [
     "CBC"
    ],
    "in_2021_2023": true,
    "note": "Table 1's EOPC and HGB rows look swapped (see notes)."
   },
   {
    "name": "WBC",
    "coding": "1000 cells/uL, continuous",
    "nhanes_variables": [
     "LBXWBCSI"
    ],
    "nhanes_files": [
     "CBC"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "HGB",
    "coding": "g/dL, continuous",
    "nhanes_variables": [
     "LBXHGB"
    ],
    "nhanes_files": [
     "CBC"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "calorie consumption",
    "coding": "kcal/day, continuous",
    "nhanes_variables": [
     "DR1TKCAL"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "Day 1 alone or a two-day mean is not stated (same as the exposure)."
   },
   {
    "name": "protein consumption",
    "coding": "gm/day, continuous",
    "nhanes_variables": [
     "DR1TPROT"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "Day 1 alone or a two-day mean is not stated (same as the exposure)."
   },
   {
    "name": "carbohydrate consumption",
    "coding": "gm/day, continuous",
    "nhanes_variables": [
     "DR1TCARB"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "Day 1 alone or a two-day mean is not stated (same as the exposure)."
   },
   {
    "name": "sugar consumption",
    "coding": "gm/day, continuous (total sugars)",
    "nhanes_variables": [
     "DR1TSUGR"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "Day 1 alone or a two-day mean is not stated (same as the exposure)."
   },
   {
    "name": "fiber consumption",
    "coding": "gm/day, continuous",
    "nhanes_variables": [
     "DR1TFIBE"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "Day 1 alone or a two-day mean is not stated (same as the exposure)."
   },
   {
    "name": "fat consumption",
    "coding": "gm/day, continuous (total fat)",
    "nhanes_variables": [
     "DR1TTFAT"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "Day 1 alone or a two-day mean is not stated (same as the exposure)."
   }
  ],
  "design": {
   "weights": "Not stated. The R survey package is named, but no weight, and no statement that the models were weighted; Table 1 and the crude ORs are unweighted (see notes).",
   "strata_psu": "not stated",
   "quote": "Statistical analysis: \"The statistical analyses were conducted utilizing R software (version 4.2.1; R Foundation for Statistical Computing; http://www.R-project.org), along with the R survey package (version 4.1–1) and Free Statistics software version 1.9.25\"",
   "software": "R 4.2.1, R survey package 4.1-1, Free Statistics software 1.9",
   "missing_data": "imputation (single imputation of missing covariates with an iterative round-robin imputer using Bayesian Ridge; complete-case analysis as a sensitivity analysis)",
   "quote_missing": "Covariates: \"A multivariate single imputation method for missing data was implemented using an iterative imputer. At each step of the round-robin imputation, a Bayesian Ridge model was used as the estimator.24\" Statistical analysis: \"First, we excluded participants with missing covariates to ascertain if the patterns of the single imputation analysis were consistent with the trends found; multivariable logistic regression modeling and RCS were employed.\""
  },
  "model": {
   "family": "logistic",
   "weighted": null,
   "quote": "Statistical analysis: \"The odds ratios (OR) and 95% confidence intervals (CIs) for the correlation between consumption of dietary zinc and asthma were calculated using logistic regression models. Sociodemographic variables such as age, sex, race, ethnicity, and PIR were taken into account while adjusting Model 1. In Model 2, additional factors, including family asthma, secondhand smoking, EOPC, WBC, and HGB, were further adjusted. The consumption of calories, protein, carbohydrates, sugar, fiber, and fat was then taken into consideration while making additional adjustments to Model 3.\""
  },
  "unstated": [
   "Day-1 recall only or the mean of two recalls, for zinc and for the energy and macronutrient covariates",
   "Whether supplement zinc was excluded (the exposure is called dietary zinc intake and supplements are not mentioned for it)",
   "Whether the regressions were survey-weighted, which weight (dietary day-1 weight WTDRD1, MEC weight, or other), how the 2017-March 2020 pre-pandemic cycle was combined with the earlier cycles, and whether strata and PSUs were used",
   "Definition of second-hand smoking (household smokers SMD460, serum cotinine, or another item)",
   "Definition of family history of asthma (presumably MCQ300b) and how missing answers were handled",
   "Lower age bound (BMDBMIC exists only for ages 2-19)",
   "Whether the zinc quartile cutpoints were weighted or unweighted (group sizes 1141, 1156, 1147, 1153 suggest unweighted sample quartiles)",
   "Whether recalls not meeting the minimum criteria (DR1DRSTZ not 1) were excluded",
   "Handling of 'Refused' and 'Don't know' for MCQ010",
   "Which variables were imputed and what entered the imputation model",
   "Whether pregnant adolescents were excluded",
   "Which level of the asthma variable the logistic model treated as the event (see notes: the printed ORs match the odds of NOT having asthma)"
  ],
  "notes": "Headline choice: the abstract's results sentence gives Q2, Q3 and Q4 vs Q1 from the fully adjusted model; following the brief's rule literally, the first one (Q2 vs Q1, OR 0.78, 0.62-0.98) is the headline. The highest-vs-lowest contrast in the same sentence is Q4 vs Q1, OR 0.71 (0.53-0.95), p = 0.022, which the Discussion also quotes; flagging in case the coordinator prefers it. PROBABLE REVERSED OUTCOME CODING: Table 1 shows asthma prevalence rising across zinc quartiles (208/1141 = 18.2%, 240/1156 = 20.8%, 243/1147 = 21.2%, 272/1153 = 23.6%; Table 1 p = 0.018), yet every Table 3 OR is below 1. Recomputed here from Table 1's counts, the unweighted crude ORs (Woolf 95% CI) vs Q1 for HAVING asthma are Q2 1.175 (0.956, 1.445), Q3 1.206 (0.981, 1.482), Q4 1.385 (1.131, 1.696); for NOT having asthma they are Q2 0.851 (0.692, 1.046), Q3 0.829 (0.675, 1.020), Q4 0.722 (0.590, 0.884), which reproduce Table 3's printed crude column (0.85 (0.69∼1.05), 0.83 (0.67∼1.02), 0.72 (0.59∼0.88)) exactly. Table 3's 'n(%)' column also lists the counts without asthma (933, 916, 904, 881). So the published ORs most likely model the absence of asthma: a re-analysis with asthma as the event should expect ORs near the reciprocals (headline 1/0.78 = 1.28, CI 1/0.98 = 1.02 to 1/0.62 = 1.61), and the paper's 'inverse association' would then be a positive one. Table 2's covariate ORs (age 0.96 per year, non-Hispanic Black 0.52 vs Mexican American, EOPC 0.9 per %) also fit an outcome of 'no asthma'. The match also shows the crude models were unweighted. Table 1's EOPC and HGB rows look swapped (EOPC '13.3 ± 1.3' %, HGB median '2.5' g/dL), and its release-cycle row reads '2014–2016' for 2015-2016. Other analyses not used for the headline: asthma attack in the past year (MCQ040, a different outcome) Q4 vs Q1 0.6 (0.37–0.99); 2011-2012 only with asthma treatment added, Q4 (≥12.15) vs Q1 (≤5.96) 0.3 (0.11–0.84); complete-case results in Supplementary Tables S1-S3 (mmc1.docx, not retrieved because the headline is in the main text). Table A's population says 'obese', but the paper includes overweight (85th to <95th percentile) and obese children and adolescents. In 2021-2023 both dietary recalls were by telephone (DR1TOT_L documentation), whereas the paper's cycles had an in-person day-1 recall. Second-hand smoking uses 'Yes' as the reference level in Table 2.",
  "adjudication": "Under the headline rule as settled, a coding with ordered categories and the lowest as reference is represented by its highest category against the reference; the extraction took the first significant contrast the abstract listed."
 },
 {
  "id": "row096",
  "rank": 53,
  "row": 96,
  "doi": "10.3389/fmed.2022.925344",
  "pmcid": "PMC9273928",
  "title": "The Association Between METS-IR and Serum Ferritin Level in United States Female: A Cross-Sectional Study Based on NHANES",
  "authors": [
   "Hao, Han",
   "Chen, Yan",
   "Xiaojuan, Ji",
   "Siqi, Zhang",
   "Hailiang, Chu",
   "Xiaoxing, Sun",
   "Qikai, Wang",
   "Mingquan, Xing",
   "Jiangzhou, Feng",
   "Hongfeng, Ge"
  ],
  "year": 2022,
  "journal": "Frontiers in Medicine",
  "table_a": {
   "predictor": "Serum ferritin levels",
   "condition": "Metabolic score for insulin resistance",
   "population": "US adult females"
  },
  "headline": {
   "abstract_quote": "There was a positive relationship between METS-IR and serum ferritin, with an effect value of (β = 0.29, 95% CI: 0.14–0.44) in a fully adjusted model adjusted for potential confounders.",
   "table_location": "Table 2 (Association between METS-IR and serum ferritin (ng/ml)), column 'Model 3, β (95% CI)', row 'METS-IR'",
   "table_quote": "| METS-IR | 0.54 (0.41, 0.68) | 0.49 (0.36, 0.63) | 0.29 (0.14, 0.44) |",
   "measure": "beta",
   "estimate": 0.29,
   "ci_low": 0.14,
   "ci_high": 0.44,
   "p_value": null,
   "exposure_contrast": "per 1-unit increase in METS-IR (continuous); outcome serum ferritin in ng/ml, untransformed, so beta is ng/ml of ferritin per METS-IR unit",
   "model_label": "Model 3 (all covariates in Table 1)",
   "covariates_in_this_model": [
    "Age (years)",
    "Race",
    "BMI (kg/m2)",
    "PIR",
    "Education",
    "Smoker",
    "Drinking",
    "Activity intensity",
    "Hypertension",
    "Diabetes",
    "Liver condition",
    "Malignant tumors",
    "Total daily energy intake (kcal)",
    "Total daily sugar intake (gm)",
    "Total daily moisture intake (gm)",
    "Total daily fat intake (gm)",
    "Total daily iron intake (mg)",
    "Ever received a blood transfusion",
    "HDL (mg/dl)",
    "FPG (mg/dl)",
    "Fasting TG (mg/dl)",
    "CRP (mg/dl)"
   ],
   "n_analytic": 4182,
   "n_quote": "Participants: \"A final total of 4,182 participants were enrolled in this study (Figure 1).\" Results: \"A total of 4,182 participants were enrolled in this study.\" Figure 1 (flowchart image): \"Final sample size (n=4182)\".",
   "events": null
  },
  "cycles": [
   "2005-2006",
   "2007-2008",
   "2009-2010",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": "20-49 (younger than 20 excluded; Discussion gives 20 to 49 years)",
   "inclusion": "Females with serum ferritin and the fasting measures needed for METS-IR (fasting glucose, fasting triglycerides, HDL-C, BMI), aged 20 or older, from NHANES 2005-2010 and 2015-2018; males from 2017-2018 excluded",
   "exclusions": [
    "Male participants from NHANES 2017-2018 (count not given)",
    "No information about serum ferritin (n=67020), leaving 17,704 (Figure 1)",
    "No information about METS-IR (n=11511), leaving 6,193 (Figure 1)",
    "Age <20 years old (n=1990) (Figure 1)",
    "No information about education (n=5), diabetes (n=4), smoking (n=1), liver condition (n=1), malignancy (n=2), hypertension (n=8), leaving 4,182 (Figure 1)"
   ],
   "quote": "Participants: \"We first excluded male participants from NAHNES 2017–2018 (only NHANES 2017–2018 contained information on serum ferritin in males). Participants included in this study were derived from NAHENS 2005–2010 and NHANES 2015–2018, because only these two time periods contained complete information on serum ferritin. A total of 84,724 participants took part in this survey during this period. We excluded participants who had no serum ferritin information, could not calculate METS-IR (n = 11,511) and were younger than 20 years old (n = 1,990). A final total of 4,182 participants were enrolled in this study (Figure 1).\" Discussion: \"Furthermore, the age of the participants included in this study fluctuated between 20 and 49 years, which leads to the possibility that our findings may not apply to other age groups\". Figure 1 (flowchart image): \"Total participants from NHANES 2005-2010 and NHANES 2015-2018 (n=84724)\"; \"Excluded: No information about serum ferritin (n=67020)\"; \"Remaining participants (n=17704)\"; \"Excluded: No information about METS-IR (n=11511)\"; \"Remaining participants (n=6193)\"; \"Excluded: Age <20 years old (n=1990) No information about: Education (n=5) Diabetes (n=4) Smoking (n=1) Liver condition (n=1) Malignancy (n=2) Hypertension (n=8)\"; \"Final sample size (n=4182)\"."
  },
  "exposure": {
   "definition": "METS-IR = ln[2 × fasting glucose (mg/dL) + fasting triglycerides (mg/dL)] × BMI (kg/m2) / ln[HDL-C (mg/dL)], continuous, per 1 unit. BMI = weight (kg) / height (m)^2.",
   "nhanes_variables": [
    "LBXGLU",
    "LBXTR (LBXTLG in 2021-2023)",
    "BMXBMI",
    "LBDHDD"
   ],
   "nhanes_files": [
    "GLU",
    "TRIGLY",
    "BMX",
    "HDL"
   ],
   "transform": "none (continuous, per 1 unit)",
   "categories": "Quartiles (secondary analysis; Table 2 labels them 'Quintiles'): \"| Q1 (18.91–31.62) | Reference | Reference | Reference |\"; Q2 (31.63–38.90); Q3 (38.91–49.05); Q4 (49.06–124.67), as in Table 2's row labels",
   "quote": "Dependent and Independent Variables: \"METS-IR as the independent variable was not directly derived from the NHANES database. METS-IR was calculated as follows: Ln [(2 × fasting glucose (mg/dL) + fasting triglycerides (mg/dL)] × body mass index (kg/m2) / {Ln [high-density lipoprotein cholesterol (mg/dL)]} (14). We obtained BMI information from examination data, calculated as BMI (kg/m2) = weight (kg)/height (m2), and all other data were collected from laboratory data.\""
  },
  "outcome": {
   "definition": "Serum ferritin (ng/ml), continuous and untransformed; 2005-2008 values (Hitachi 912) converted to Elecsys 170 equivalents with the Deming equation log10(E170) = 0.989 × log10(Hitachi 912) + 0.049",
   "nhanes_variables": [
    "LBXFER"
   ],
   "nhanes_files": [
    "FERTIN"
   ],
   "quote": "Dependent and Independent Variables: \"Serum ferritin (ng/ml) as the dependent variable was derived from laboratory data. In NHANES 2004–2008, serum ferritin concentrations were measured by the Roche Tina-quant serum ferritin immunoturbidimetric method on a Hitachi 912 clinical analyser (Roche Diagnostics, Basel, Switzerland). As the Hitachi 912 clinical analyser was discontinued by the manufacturer in 2009, NHANES 2009–2010 and NHANES 2015–2018 used the Roche Elecsys 170 clinical analyser to measure serum ferritin concentrations. However, the NHANES working group converted this difference based on the Deming equation (Log10 (E170) = 0.989*Log10(Hitachi 912) + 0.049).\" Discussion: \"we have used the recommended Deming regression for conversion\""
  },
  "covariates": [
   {
    "name": "Age (years)",
    "coding": "continuous",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "Race",
    "coding": "Mexican American, White, Black, Other race (Table 1)",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Other race presumably merges Other Hispanic and Other/multiracial; not stated."
   },
   {
    "name": "BMI (kg/m2)",
    "coding": "continuous",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Also a component of METS-IR."
   },
   {
    "name": "PIR",
    "coding": "continuous (Table 1 mean ± SD)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Missing values presumably mean-imputed under the 10% rule; not stated."
   },
   {
    "name": "Education",
    "coding": "Less than high school, High school, More than high school (Table 1)",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "Smoker",
    "coding": "Yes/No: smoked at least 100 cigarettes in lifetime",
    "nhanes_variables": [
     "SMQ020"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "Drinking",
    "coding": "Yes, No, Unclear: consumed alcohol at least 12 times in a year",
    "nhanes_variables": [
     "ALQ101",
     "ALQ121"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "Item not named. The 2005-2016 alcohol files asked ALQ101 (at least 12 drinks in any one year), which 2021-2023 lacks; the paper's wording ('at least 12 times in a year') can be built from ALQ121 in ALQ_L (past-12-month frequency, at least once a month). How 2017-2018 (no ALQ101) was coded is not stated."
   },
   {
    "name": "Activity intensity",
    "coding": "Mild, Moderate, Vigorous, from the intensity of activity at work and recreation",
    "nhanes_variables": [
     "PAQ605",
     "PAQ620",
     "PAQ650",
     "PAQ665"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": false,
    "note": "Derivation rule not stated (GPAQ items listed are those of 2007-2018; 2005-2006 used different items). PAQ_L in 2021-2023 has leisure-time moderate and vigorous activity (PAD790Q/U, PAD810Q/U) but no work activity."
   },
   {
    "name": "Hypertension",
    "coding": "Yes/No from the questionnaire (affirmative response)",
    "nhanes_variables": [
     "BPQ020"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "Diabetes",
    "coding": "Yes, No, Borderline (Table 1)",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "Liver condition",
    "coding": "Yes/No",
    "nhanes_variables": [
     "MCQ160l"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "Malignant tumors",
    "coding": "Yes/No",
    "nhanes_variables": [
     "MCQ220"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "Total daily energy intake (kcal)",
    "coding": "<1,863.94; ≥1,863.94; Unclear",
    "nhanes_variables": [
     "DR1TKCAL",
     "DR2TKCAL"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "Mean of day 1 and day 2 recalls (Methods); cut at the Table 1 value with missing kept as 'Unclear'."
   },
   {
    "name": "Total daily sugar intake (gm)",
    "coding": "<112.20; ≥ 112.20; Unclear",
    "nhanes_variables": [
     "DR1TSUGR",
     "DR2TSUGR"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "Mean of day 1 and day 2 recalls (Methods); cut at the Table 1 value with missing kept as 'Unclear'."
   },
   {
    "name": "Total daily moisture intake (gm)",
    "coding": "<68.69; ≥ 68.69; Unclear",
    "nhanes_variables": [
     "DR1TMOIS",
     "DR2TMOIS"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "Mean of day 1 and day 2 recalls (Methods); cut at the Table 1 value with missing kept as 'Unclear'. Table 1's moisture rows repeat the fat rows' counts exactly, and 68.69 g is implausible as a moisture cutpoint, so one of the two rows is probably mislabeled."
   },
   {
    "name": "Total daily fat intake (gm)",
    "coding": "<73.01; ≥73.01; Unclear",
    "nhanes_variables": [
     "DR1TTFAT",
     "DR2TTFAT"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "Mean of day 1 and day 2 recalls (Methods); cut at the Table 1 value with missing kept as 'Unclear'."
   },
   {
    "name": "Total daily iron intake (mg)",
    "coding": "<13.95; ≥13.95; Unclear",
    "nhanes_variables": [
     "DR1TIRON",
     "DR2TIRON"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "Mean of day 1 and day 2 recalls (Methods); cut at the Table 1 value with missing kept as 'Unclear'."
   },
   {
    "name": "Ever received a blood transfusion",
    "coding": "Yes, No, Unclear",
    "nhanes_variables": [
     "MCQ092"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": false,
    "note": "MCQ092 is not in MCQ_L (2021-2023)."
   },
   {
    "name": "HDL (mg/dl)",
    "coding": "continuous",
    "nhanes_variables": [
     "LBDHDD"
    ],
    "nhanes_files": [
     "HDL"
    ],
    "in_2021_2023": true,
    "note": "Also a component of METS-IR."
   },
   {
    "name": "FPG (mg/dl)",
    "coding": "continuous",
    "nhanes_variables": [
     "LBXGLU"
    ],
    "nhanes_files": [
     "GLU"
    ],
    "in_2021_2023": true,
    "note": "Also a component of METS-IR."
   },
   {
    "name": "Fasting TG (mg/dl)",
    "coding": "continuous",
    "nhanes_variables": [
     "LBXTR"
    ],
    "nhanes_files": [
     "TRIGLY"
    ],
    "in_2021_2023": true,
    "note": "Also a component of METS-IR; LBXTLG in TRIGLY_L (2021-2023)."
   },
   {
    "name": "CRP (mg/dl)",
    "coding": "continuous",
    "nhanes_variables": [
     "LBXCRP",
     "LBXHSCRP"
    ],
    "nhanes_files": [
     "CRP",
     "HSCRP"
    ],
    "in_2021_2023": true,
    "note": "2005-2010 CRP (LBXCRP, mg/dL); 2015-2018 hs-CRP (LBXHSCRP, mg/L); 2021-2023 has hs-CRP LBXHSCRP (mg/L) in HSCRP_L, divide by 10 for mg/dl. Harmonization not stated."
   }
  ],
  "design": {
   "weights": "'2-year sample weights'; which one (MEC exam or fasting subsample) and how they were combined across the five cycles is not stated.",
   "strata_psu": "not stated",
   "quote": "Statistical Analysis: \"In order to make the NHANES data more representative of the whole United States cohort, we used 2-year sample weights in this study.\"",
   "software": "R and EmpowerStats. Statistical Analysis: \"All data extraction and analysis were performed in R (http://www.R-project.org) and EmpowerStats (http://www.empowerstats.com).\"",
   "missing_data": "mean imputation for continuous variables with up to 10% missing; otherwise grouped with a separate 'unclear' category; categorical variables with more than 10 missing get an 'unclear' category, otherwise those participants are deleted",
   "quote_missing": "Statistical Analysis: \"If the missing data for a continuous variable was within 10%, we would use the average of that variable instead. Otherwise, continuous variables would be grouped according to specific rules and the missing data would be set as a separate “unclear group”. If more than 10 samples were missing for a categorical variable, the missing data will be grouped separately as the “unclear group”, otherwise they would be deleted.\""
  },
  "model": {
   "family": "linear",
   "weighted": true,
   "quote": "Statistical Analysis: \"In order to make the NHANES data more representative of the whole United States cohort, we used 2-year sample weights in this study.\" ... \"Multiple linear regression analysis was used in the analysis to explore the relationship between METS-IR and serum ferritin. Depending on the adjustment for covariates, three models were generated. Model 1: no adjustment for covariates; Model 2: age and race were adjusted; Model 3: all covariates shown in Table 1 were adjusted.\""
  },
  "unstated": [
   "Which 2-year weight (MEC exam or fasting subsample) and how weights were combined across the five cycles; whether strata and PSUs were used",
   "How women aged 50 and older from 2017-2018 (when ferritin was measured in both sexes aged 12+) were excluded, given the stated 20-49 range",
   "Whether pregnant women were excluded",
   "Whether a fasting-duration criterion was applied beyond membership in the fasting subsample",
   "Which cycles' ferritin values were Deming-converted (the paper names the Hitachi 912 for 2004-2008)",
   "The 'specific rules' for grouping the diet variables (only the cutpoints are given) and how 'Unclear' was assigned",
   "How the four race groups were built from RIDRETH1",
   "How 'Activity intensity' (mild, moderate, vigorous) was derived from work and recreation items, including in 2005-2006 before the GPAQ items",
   "Which alcohol item defined 'at least 12 times in a year' in each cycle",
   "How CRP (2005-2010, mg/dL) and hs-CRP (2015-2018, mg/L) were put on one scale",
   "Which continuous variables were mean-imputed",
   "How the diet means were formed when only day 1 was available"
  ],
  "notes": "Direction: Suchak et al.'s labels make serum ferritin the predictor and METS-IR the condition, but the paper models serum ferritin (ng/ml) as the dependent variable and METS-IR as the independent one (Abstract: \"We used METS-IR and serum ferritin as the independent and dependent variables in this study\"); the headline beta is ng/ml of ferritin per METS-IR unit. Table A's NHANES dates (2005-2010) omit the 2015-2016 and 2017-2018 cycles the paper also used. Model 3 adjusts for all Table 1 covariates, which include BMI, HDL, FPG and fasting TG, the components of METS-IR. Table 2 labels its METS-IR groups 'Quintiles' but shows four; the text calls them quartiles. Figure 1 (flowchart image, saved as scratchpad/dl/supp/row096/Fig1.jpg) supplies the exclusion counts; its starting total (n=84724) is larger than the five cycles' combined samples (each roughly 9,000 to 10,500 people), and it shows no step for the 2017-2018 males, so the flow counts may be unreliable. Ferritin eligibility by cycle: 2017-2018 measured both sexes aged 12+ (FERTIN_J documentation, saved as scratchpad/dl/supp/row096/FERTIN_J.htm), hence the paper's male exclusion; 2021-2023 measures only children 1-5 and females 12-49 (FERTIN_L), so a replication sample of women 20-49 matches the paper's stated range. Assay: the paper names the Elecsys 170 for 2009-2010 and 2015-2018 and the Hitachi 912 for 2004-2008; NHANES documentation lists the Roche Cobas e601 for 2017-2018 and 2021-2023. Same construct and units (ng/mL). No p-value is printed for the headline estimate. Abstract and Conclusions call the association more pronounced at BMI < 24.9 kg/m2, while Table 2's Model 3 estimate for BMI <24.9 is 0.62 (−0.12, 1.37) and the Results name BMI ≥ 24.9; this does not affect the headline. Author names are recorded as the journal metadata gives them; for authors 3-10 the given and family names appear inverted (likely surnames Ji, Zhang, Chu, Sun, Wang, Xing, Feng, Ge).",
  "adjudication": null
 },
 {
  "id": "row131",
  "rank": 55,
  "row": 131,
  "doi": "10.3389/fnut.2022.936926",
  "pmcid": "PMC9253671",
  "title": "Association of Dietary Fiber Intake With Myocardial Infarction and Stroke Events in US Adults: A Cross-Sectional Study of NHANES 2011–2018",
  "authors": [
   "Dong, Weiwei",
   "Yang, Zhiyong"
  ],
  "year": 2022,
  "journal": "Frontiers in Nutrition",
  "table_a": {
   "predictor": "Dietary fiber intake",
   "condition": "Stroke",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": null,
   "table_location": "Table 3, Model 3, row Tertile3",
   "table_quote": "| Fiber intake (g/day) | 0.98 (0.97, 0.99) | 0.97 (0.96, 0.99) | 0.98 (0.96, 1.00) |",
   "measure": "OR",
   "estimate": 0.64,
   "ci_low": 0.46,
   "ci_high": 0.91,
   "p_value": "not reported",
   "exposure_contrast": "tertile 3 vs tertile 1 of dietary fiber intake",
   "model_label": "Model 3",
   "covariates_in_this_model": [
    "age",
    "sex",
    "race",
    "marital status",
    "educational level",
    "PIR",
    "BMI",
    "smoking status",
    "systolic blood pressure",
    "diastolic blood pressure",
    "glucose level",
    "cholesterol level",
    "triglyceride level",
    "HDL level",
    "glycohemoglobin level",
    "energy intake",
    "vigorous activity",
    "diabetes",
    "hypertension",
    "hypercholesterolemia",
    "sleeping disorder",
    "hypoglycemic drugs",
    "antihypertensive drugs",
    "lipid-lowing drugs",
    "aspirin drugs"
   ],
   "n_analytic": 8872,
   "n_quote": "Results: \"A total of 8,872 participants were recruited in this study\"; Figure 1 (image) final box: \"N=8872\"",
   "events": null
  },
  "cycles": [
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=18 as stated (in practice >=20, see notes)",
   "inclusion": "NHANES 2011-2018 participants aged 18 or over with complete data on the key analysis variables",
   "exclusions": [
    "Age<18 N=15331 (from N=39156 to N=23825)",
    "PIR unavailable or missing (N=2528)",
    "Educational level unavailable or missing (N=1071)",
    "Material [marital] status unavailable or missing (N=6); N=20220 remain",
    "BMI unavailable or missing (N=993)",
    "Blood pressure unavailable or missing (N=2008)",
    "Glucose/Cholesterol/Triglycerides/HDL/Glycohemoglobin unavailable or missing (N=996); N=16223 remain",
    "Vigorous activity unavailable or missing (N=6590)",
    "Diabetes/hypertension/Hypercholesteremia status unavailable or missing (N=150)",
    "Sleep disorder unavailable or missing (N=1); N=9482 remain",
    "Stroke/MI status unavailable or missing (N=15)",
    "Fiber intake data unavailable or missing (N=595); N=8872 analyzed"
   ],
   "quote": "Methods, Study Population: \"Individuals were excluded if they were younger than 18 years (n = 15,331) or responded with missing values for key analysis variables.\" Figure 1 (image, transcribed): \"NHANES 2011-2018 (N=39156) | Age<18 N=15331 | N=23825 | Demographic: PIR unavailable or missing (N=2528); Educational level unavailable or missing (N=1071); Material status unavailable or missing (N=6) | N=20220 | Examination & Laboratory: BMI unavailable or missing (N=993); Blood pressure unavailable or missing (N=2008); Glucose/Cholesterol/Triglycerides/HDL/Glycohemoglobin unavailable or missing (N=996) | N=16223 | Questionnaire: Vigorous activity unavailable or missing (N=6590); Diabetes/hypertension/Hypercholesteremia status unavailable or missing (N=150); Sleep disorder unavailable or missing (N=1) | N=9482 | Stroke/MI status unavailable or missing (N=15); Fiber intake data unavailable or missing (N=595) | N=8872\""
  },
  "exposure": {
   "definition": "Total dietary fiber intake (g/day) from the first (day-1, in-person at the MEC) 24-h dietary recall, summing food and supplemental sources; continuous per 1 g/day.",
   "nhanes_variables": [
    "DR1TFIBE",
    "DS1TFIBE (day-1 supplement recall, 2011-2018; source not named in the paper)",
    "DSQTFIBE (30-day supplement file; the only supplement fiber in 2021-2023)"
   ],
   "nhanes_files": [
    "DR1TOT_G",
    "DR1TOT_H",
    "DR1TOT_I",
    "DR1TOT_J",
    "DS1TOT_G-J (or DSQTOT_G-J)",
    "DR1TOT_L and DSQTOT_L (2021-2023)"
   ],
   "transform": "none (continuous g/day); tertiles in secondary models (cutpoints not reported; tertile means 7.51, 14.90 and 28.79 g/day)",
   "categories": null,
   "quote": "Methods, Exposure and Outcome Definitions: \"Similar to earlier studies (18, 19), this study used the first 24-h dietary recall conducted by trained food recall data collectors at the MEC.\" and \"The total intake of fiber was calculated by summing the amounts from food and supplemental sources.\""
  },
  "outcome": {
   "definition": "Self-reported doctor-diagnosed heart attack (MI) and/or stroke, called nonfatal cardiovascular/cerebrovascular events; event = yes to either question. The paper reports no stroke-only estimate.",
   "nhanes_variables": [
    "MCQ160E",
    "MCQ160F"
   ],
   "nhanes_files": [
    "MCQ_G",
    "MCQ_H",
    "MCQ_I",
    "MCQ_J",
    "MCQ_L (2021-2023)"
   ],
   "quote": "Methods, Exposure and Outcome Definitions: \"We determined the outcomes using a Medical Condition Questionnaire. When a participant answered “yes” to the question “has a doctor ever told you that you had a heart attack,” we considered that he or she had MI. Similarly, when a participant answered “yes” to the question “has a doctor ever told you that you had a stroke,” we considered that he or she had a stroke (20, 21). Self-reported stroke and MI measures have been used in previous epidemiological studies using NHANES data, and results of several studies have revealed the self-reported measurement method is reliable (21–24). Outcomes included patient-reported nonfatal MI and/or stroke.\""
  },
  "covariates": [
   {
    "name": "age",
    "coding": "continuous years (Table 2)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "sex",
    "coding": "men/women",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "race",
    "coding": "Mexican American, other races, non-Hispanic White, non-Hispanic Black",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "other races = Other Hispanic plus Other Race including multiracial"
   },
   {
    "name": "marital status",
    "coding": "married/living with partner, widowed/divorced/separated, never married",
    "nhanes_variables": [
     "DMDMARTL (2011-2018)",
     "DMDMARTZ (2021-2023)"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DMDMARTZ has exactly these three groups"
   },
   {
    "name": "educational level",
    "coding": "< high school, high school, > high school",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "PIR",
    "coding": "<1.2 or >=1.2",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "BMI",
    "coding": "<25, 25-30, >=30 kg/m2",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "smoking status",
    "coding": "current smoker yes, no, or unknown",
    "nhanes_variables": [
     "unstated (SMQ040 is the likely item)"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "weighted Unknown 86.80% (Table 2), so the item used is unclear"
   },
   {
    "name": "systolic blood pressure",
    "coding": "mean of three readings, mmHg",
    "nhanes_variables": [
     "BPXSY1-BPXSY3 (2011-2018)",
     "BPXOSY1-BPXOSY3 (2021-2023)"
    ],
    "nhanes_files": [
     "BPX",
     "BPXO_L"
    ],
    "in_2021_2023": true,
    "note": "2021-2023 readings are oscillometric, not auscultatory"
   },
   {
    "name": "diastolic blood pressure",
    "coding": "mean of three readings, mmHg",
    "nhanes_variables": [
     "BPXDI1-BPXDI3 (2011-2018)",
     "BPXODI1-BPXODI3 (2021-2023)"
    ],
    "nhanes_files": [
     "BPX",
     "BPXO_L"
    ],
    "in_2021_2023": true,
    "note": "oscillometric in 2021-2023"
   },
   {
    "name": "glucose level",
    "coding": "mg/dl, continuous",
    "nhanes_variables": [
     "unstated: LBXSGL (biochemistry) or LBXGLU (fasting)"
    ],
    "nhanes_files": [
     "BIOPRO or GLU"
    ],
    "in_2021_2023": true,
    "note": "the small loss for missing labs (996) suggests the non-fasting biochemistry value"
   },
   {
    "name": "cholesterol level",
    "coding": "mg/dl, continuous",
    "nhanes_variables": [
     "LBXTC (or LBXSCH)"
    ],
    "nhanes_files": [
     "TCHOL (or BIOPRO)"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "triglyceride level",
    "coding": "mg/dl, continuous",
    "nhanes_variables": [
     "unstated: LBXSTR (biochemistry) or LBXTR/LBXTLG (fasting)"
    ],
    "nhanes_files": [
     "BIOPRO or TRIGLY"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "HDL level",
    "coding": "mg/dl, continuous",
    "nhanes_variables": [
     "LBDHDD"
    ],
    "nhanes_files": [
     "HDL"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "glycohemoglobin level",
    "coding": "%, continuous",
    "nhanes_variables": [
     "LBXGH"
    ],
    "nhanes_files": [
     "GHB"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "energy intake",
    "coding": "kcal/day, continuous",
    "nhanes_variables": [
     "DR1TKCAL"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "vigorous activity",
    "coding": "yes/no; work activity with large increases in breathing or heart rate (carrying or lifting heavy loads, digging or construction work) for at least 10 min continuously",
    "nhanes_variables": [
     "PAQ605"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": false,
    "note": "PAQ_L has leisure-time activity only (no work activity)"
   },
   {
    "name": "diabetes",
    "coding": "yes, no, or borderline",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "hypertension",
    "coding": "yes or no",
    "nhanes_variables": [
     "unstated (BPQ020 likely)"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "hypercholesterolemia",
    "coding": "yes or no",
    "nhanes_variables": [
     "unstated (BPQ080 likely)"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "sleeping disorder",
    "coding": "yes or no",
    "nhanes_variables": [
     "unstated (SLQ050 likely)"
    ],
    "nhanes_files": [
     "SLQ"
    ],
    "in_2021_2023": false,
    "note": "SLQ_L has no trouble-sleeping or sleep-disorder question"
   },
   {
    "name": "hypoglycemic drugs",
    "coding": "yes, no, or unknown",
    "nhanes_variables": [
     "unstated (DIQ070 and/or DIQ050)"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "antihypertensive drugs",
    "coding": "yes, no, or unknown",
    "nhanes_variables": [
     "unstated (BPQ050A in 2011-2018; BPQ150 in 2021-2023)"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "lipid-lowing drugs",
    "coding": "yes, no, or unknown",
    "nhanes_variables": [
     "unstated (BPQ100D in 2011-2018; BPQ101D in 2021-2023)"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "aspirin drugs",
    "coding": "preventive aspirin use yes, no, or unknown",
    "nhanes_variables": [
     "unstated (RXQ515)"
    ],
    "nhanes_files": [
     "RXQASA"
    ],
    "in_2021_2023": true,
    "note": "asked at ages 40+ only, which explains a large 'unknown' group"
   }
  ],
  "design": {
   "weights": "unstated: 'an appropriate NHANES sample weight' (which weight, e.g. WTDRD1 or WTMEC2YR, and the four-cycle divisor are not given)",
   "strata_psu": "complex multistage cluster design accounted for; strata and PSU variables not named",
   "quote": "Methods, Statistical Analyses: \"All statistical analyses were performed based on the Center for Disease Control and Prevention guideline. A complex multistage cluster surgery design analysis was considered, and an appropriate NHANES sample weight was applied.\"",
   "software": "EmpowerStats (www.empowerstats.com) and R 3.4.3",
   "missing_data": "complete case for the key variables (Figure 1), with 'unknown' categories kept for smoking and the four medication variables",
   "quote_missing": "Methods, Study Population: \"Individuals were excluded if they were younger than 18 years (n = 15,331) or responded with missing values for key analysis variables.\""
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "Methods, Statistical Analyses: \"Univariate and multivariate logistic regression analyses were used to explore the association between fiber intake and nonfatal cardiovascular/cerebrovascular events. In multivariate logistic regression, model 1 was adjusted for no covariates; model 2 was adjusted for age, sex, and race; and model 3 was adjusted for all covariates.\" and \"We performed all the analyses using Empower software (www.empowerstats.com; X&Y solutions, Inc., Boston, MA, USA) and R version 3.4.3\""
  },
  "unstated": [
   "Which sample weight (WTDRD1 or WTMEC2YR) and the four-cycle divisor",
   "Whether supplement fiber came from the day-1 supplement recall (DS1TFIBE) or the 30-day questionnaire (DSQTFIBE), and how non-users were coded",
   "Whether unreliable dietary recalls (DR1DRSTZ not 1) were excluded",
   "The NHANES items behind current smoker (yes/no/unknown), hypertension, hypercholesterolemia, sleep disorder and the four medication variables",
   "Whether glucose and triglycerides are fasting or non-fasting (biochemistry) values, and which total cholesterol measure",
   "How continuous covariates entered the model",
   "Whether pregnant women were excluded",
   "Tertile cutpoints (only tertile means are given)",
   "The p-value of the headline estimate (not printed)"
  ],
  "notes": "Table A's condition label is 'Stroke', but the paper's only outcome is the composite of self-reported MI and/or stroke; no stroke-only estimate is reported. The abstract (taken from the Europe PMC XML, since the text file's abstract section is empty) gives no effect estimate with a 95% CI (\"Higher fiber intake indicated a stable negative association with nonfatal cardiovascular/cerebrovascular events in the multivariate logistic regression analysis, weighted generalized additive model, and smooth curve fitting.\"), so the headline is the most-adjusted continuous estimate in the main regression table (Table 3, Model 3). Age: stated >= 18, but education (DMDEDUC2), marital status and the MCQ160 items are asked only at ages 20+; the 1,071 excluded for missing education about equals the 1,208 participants aged 18-19 (23,825 aged 18+ in Figure 1 minus 22,617 aged 20+ per the DEMO_G to DEMO_J documentation; some were already lost to missing PIR), so the analytic sample is probably all aged 20+. Current smoker is odd (weighted Yes 6.09%, No 7.11%, Unknown 86.80%, Table 2). Vigorous activity is the work item and was missing for 6,590 people (Figure 1); with sleep disorder it is one of two Model 3 covariates that 2021-2023 cannot supply. 2021-2023 changes for covariates: blood pressure is oscillometric (BPXO_L); BPQ150 replaces BPQ050A and BPQ101D replaces BPQ100D; DMDMARTZ replaces DMDMARTL. The Discussion misstates the direction (\"higher dietary fiber intake was independently associated with an increased prevalence of stroke and MI\"). Analyses were run in EmpowerStats. The exclusion of 131 extreme intakes applies only to the sensitivity analysis (Supplementary Table S1, saved under scratchpad/dl/supp/row131/).",
  "adjudication": "The abstract reports no estimate with a 95% CI. The main regression table's most-adjusted continuous estimate, OR 0.98 (0.96, 1.00) per g/day, has no p-value and a CI that does not exclude 1, so it is not significant; the headline is the table's most-adjusted highest-versus-reference estimate."
 },
 {
  "id": "row087",
  "rank": 57,
  "row": 87,
  "doi": "10.3389/fpubh.2024.1280163",
  "pmcid": "PMC10904630",
  "title": "Association between blood heavy metal exposure levels and risk of metabolic dysfunction associated fatty liver disease in adults: 2015–2020 NHANES large cross-sectional study",
  "authors": [
   "Tang, Song",
   "Luo, Simin",
   "Wu, Zhendong",
   "Su, Jiandong"
  ],
  "year": 2024,
  "journal": "Frontiers in Public Health",
  "table_a": {
   "predictor": "Heavy metals exposure",
   "condition": "Metabolic-associated fatty liver conditions",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "The OR (95% CI) for MAFLD prevalence was 3.936 (2.631–5.887) for every 1 unit increase in Log Mn until serum Mn levels rose to the turning point (Log Mn = 1.10, Mn = 12.61 μg/L).",
   "table_location": "Table 3 (Threshold effect analysis of blood heavy metals on MAFLD using piecewise linear regression), block 'Mn', row 'Log Mn < 1.10', column 'OR (95% CI)'",
   "table_quote": "| Log Mn < 1.10 | 3.936 (2.631–5.887) | <0.001 |",
   "measure": "OR",
   "estimate": 3.936,
   "ci_low": 2.631,
   "ci_high": 5.887,
   "p_value": "<0.001",
   "exposure_contrast": "per 1-unit increase in log10 whole-blood Mn (a tenfold increase), within the segment log10 Mn < 1.10 (Mn below 12.61 ug/L as printed; 10^1.10 = 12.59); knot (turning point) at log10 Mn = 1.10",
   "model_label": "Threshold effect (piecewise) logistic model, segment 'Log Mn < 1.10' (Table 3)",
   "covariates_in_this_model": [],
   "n_analytic": 8542,
   "n_quote": "Results: \"After excluding those lacking MAFLD, heavy metal blood level data (with the exceptions mentioned above), 8,542 adult participants (age ≥ 20 years) were included in the analysis (Figure 1).\" The segment below the knot is not sized in the paper (7,131 participants in the authors' data, see notes).",
   "events": 4370
  },
  "cycles": [
   "2015-2016",
   "2017-2018",
   "2017-March 2020 pre-pandemic"
  ],
  "population": {
   "age": ">=20 as stated; in fact 40-80 in the analytic sample (see notes)",
   "inclusion": "NHANES 2015-2020 participants aged 20 or over with MAFLD status and blood levels of all 10 metals (Pb, Hg, Cd, Mn, Se, Cr, Co, inorganic, methyl and ethyl mercury)",
   "exclusions": [
    "Participants with age <20 (n=14265) (Figure 1)",
    "Participants with missing data about MAFLD (n=7729) (Figure 1)",
    "Participants with missing data on blood heavy metal level (n=4249) (Figure 1)"
   ],
   "quote": "Results: \"Survey data were collected from 34,785 participants. After excluding those lacking MAFLD, heavy metal blood level data (with the exceptions mentioned above), 8,542 adult participants (age ≥ 20 years) were included in the analysis (Figure 1).\" Figure 1 (image): \"Participants enrolled in analyses (n =8542)\""
  },
  "exposure": {
   "definition": "Whole-blood manganese (ug/L), measured by mass spectrometry after dilution, log10-transformed; the headline is the log-odds slope for log10 Mn in participants below the turning point log10 Mn = 1.10.",
   "nhanes_variables": [
    "LBXBMN"
   ],
   "nhanes_files": [
    "PBCD_I",
    "PBCD_J",
    "P_PBCD",
    "PBCD_L (2021-2023)"
   ],
   "transform": "log10; two segments split at log10 Mn = 1.10 (Mn = 12.61 ug/L as printed)",
   "categories": null,
   "quote": "Methods, Exposure: \"Blood levels of 10 different metals [plumbum (Pb), hydrargyrum (Hg), cadmium (Cd), Mn, Se, chromium (Cr), cobalt (Co), inorganic hydrargyrum (InHg), methyl hydrargyrum (MeHg) and ethyl hydrargyrum (EtHg)] were obtained by direct extraction of participant laboratory data.\" and \"focuses on the direct measurement of heavy metals in whole blood samples using mass spectrometry after a simple dilution sample preparation procedure.\" Methods, Data analysis: \"serum heavy metal concentrations were transformed on a Log10 scale.\""
  },
  "outcome": {
   "definition": "MAFLD = hepatic steatosis of any grade (assessment method not stated; the authors' data match CAP >= 269 dB/m, see notes) plus metabolic dysfunction: BMI >= 25 kg/m2; or type 2 diabetes (antidiabetic drug use, fasting glucose >= 7.0 mmol/L, HbA1c > 6.4%, or OGTT); or at least two of: waist > 102 cm (men) / > 88 cm (women); BP >= 130/85 mmHg or antihypertensive drugs; triglycerides >= 1.70 mmol/L or lipid-lowering drugs; HDL-C < 1.0 mmol/L (men) / < 1.3 mmol/L (women) or lipid-lowering drugs; prediabetes (fasting glucose 5.6-6.9 mmol/L, HbA1c 5.7-6.4%, or OGTT); HOMA-IR >= 2.5; CRP > 2 mg/L. Participants with missing diagnosis-related data were excluded. Variable names below are a mapping; the paper names none.",
   "nhanes_variables": [
    "LUXCAPM",
    "BMXBMI",
    "BMXWAIST",
    "DIQ050",
    "DIQ070",
    "LBXGLU",
    "LBXGH",
    "BPXSY1-3/BPXDI1-3 (BPX_J) or BPXOSY1-3/BPXODI1-3 (P_BPXO, BPXO_L)",
    "BPQ050A (2017-2020) / BPQ150 (2021-2023)",
    "LBXTR (2017-2020) / LBXTLG (2021-2023)",
    "BPQ100D (2017-2020) / BPQ101D (2021-2023)",
    "LBDHDD",
    "LBXIN",
    "LBXHSCRP"
   ],
   "nhanes_files": [
    "LUX",
    "BMX",
    "DIQ",
    "GLU",
    "GHB",
    "BPX/BPXO",
    "BPQ",
    "TRIGLY",
    "HDL",
    "INS",
    "HSCRP"
   ],
   "quote": "Methods, Outcome: \"The primary outcome of this study was the presence or absence of MAFLD, which was defined as the combination of steatosis, irrespective of the gradation, and metabolic dysfunction. This was characterized by either overweight (Body mass index (BMI) ≥25 kg/m2), type 2 diabetes mellitus defined as antidiabetic drug use, fasting plasma glucose ≥7.0 mmol/L, HbA1c > 6.4% or based on oral glucose tolerance test (OGTT), or a combination of at least two of the following metabolic abnormalities: (1) waist circumference > 102 cm for male and > 88 cm for female; (2) blood pressure ≥ 130/85 mmHg or antihypertensive drug use; (3) plasma triglycerides ≥1.70 mmol/L or lipid-lowering drug treatment; (4) high-density lipoprotein cholesterol (HDL-C) <1.0 mmol/L for men and < 1.3 mmol/L for women or lipid-lowering drug treatment; (5) prediabetes defined as fasting plasma glucose 5.6–6.9 mmol/L, HbA1c 5.7–6.4% or matching OGTT; (6) homeostatic model assessment of insulin resistance (HOMA-IR) of ≥2.5; (7) or C-reactive protein (CRP) level > 2 mg/L (10). Those who met the diagnostic criteria were identified as MAFLD patients, those who did not were identified as controls, and those with missing diagnosis-related data were removed.\""
  },
  "covariates": [],
  "design": {
   "weights": "none: the paper never mentions weights, strata or PSUs; the authors' Data_Sheet_1.CSV has wtint2yr missing for all 15,560 rows labeled 2019-2020 (5,273 of the 8,542 analytic rows), and an unweighted fit reproduces Table 3 exactly",
   "strata_psu": "not used (not mentioned)",
   "quote": "Methods, Data analysis: \"All data analyzes were conducted using R.3.5.2/R4.2.2.2 Sample sizes were based on available data and no ex ante sample size calculations were performed.\"",
   "software": "R 3.5.2 / R 4.2.2",
   "missing_data": "complete case",
   "quote_missing": "Methods, Outcome: \"those with missing diagnosis-related data were removed.\" Results: \"After excluding those lacking MAFLD, heavy metal blood level data (with the exceptions mentioned above), 8,542 adult participants (age ≥ 20 years) were included in the analysis (Figure 1).\""
  },
  "model": {
   "family": "logistic; piecewise, fitted separately below and above the turning point (see notes)",
   "weighted": false,
   "quote": "Methods, Data analysis: \"Additionally, a piecewise linear regression model was applied to examine the threshold effect of serum heavy metal concentrations on the risk of MAFLD using a smoothing function. Threshold levels (i.e., turning points) were determined by iterative trials involving the selection of turning points along predefined intervals, followed by the selection of turning points that gave the maximum model likelihood.\""
  },
  "unstated": [
   "How steatosis was assessed and its cutoff (the paper says only 'steatosis, irrespective of the gradation'); the authors' data match CAP >= 269 dB/m with no exam-quality filter (inference, see notes)",
   "Which NHANES files were combined (the paper says 2015 to 2020; the data show 2015-2016, 2017-2018 and the 2017-March 2020 pre-pandemic files stacked)",
   "Covariates in the threshold model (none named; the refit shows none)",
   "Survey weights, strata and PSUs (none mentioned)",
   "Whether the two segments share an intercept or are fitted separately (the refit shows separate fits)",
   "The interval searched for the turning point",
   "Which variables define antidiabetic, antihypertensive and lipid-lowering drug use (questionnaire items or the prescription file)",
   "Whether triglycerides, glucose and HOMA-IR came from the fasting subsample, and how participants without fasting data were classified",
   "Handling of metal values below the limit of detection (the data hold NHANES fill values, e.g. Cr 0.29)",
   "That requiring chromium and cobalt restricts the sample to ages 40 and over",
   "Whether pregnant women were excluded"
  ],
  "notes": "Checked against the authors' supplementary dataset (Data_Sheet_1.CSV, 34,785 rows, from the Europe PMC supplementary files, saved under scratchpad/dl/supp/row087/); it reproduces Figure 1 exactly (age <20: 14,265; missing MAFLD: 7,729; missing any of the 10 metals: 4,249; final 8,542 with 4,172 non-MAFLD and 4,370 MAFLD) and the Table 2 medians (Mn 8.92 vs 9.18; Se 183.52 vs 188.11). (1) Duplicated participants: the 2017-March 2020 pre-pandemic files already contain the 2017-2018 participants under new SEQNs, and the authors stacked them with the 2017-2018 files. Every one of the 7,513 rows labeled 2017-2018 with Mn and Se has a row labeled 2019-2020 with the same sex, age, race, Mn and Se (control: none of the 4,987 such 2015-2016 rows does), and 3,141 of the 3,269 analytic rows labeled 2017-2018 have an exact twin (sex, age, race and eight metal values) among the 5,273 analytic rows labeled 2019-2020, so the 8,542 rows hold about 5,400 distinct people. (2) No 2015-2016 participant has MAFLD data (no elastography before 2017), so the analytic sample is 2017-2020 only. (3) Requiring chromium and cobalt, measured only at ages 40 and over (ages of rows with Cr: 40 to 80), limited the analytic sample to ages 40-80 (Table 1 medians 60 and 60); Cr was measured for 99.8% (2017-2018) and 99.6% (rows labeled 2019-2020) of participants aged 40+ with blood Mn, so the requirement amounts to an age >= 40 restriction. (4) Steatosis: merging the 2017-2018 rows with LUX_J (2017-2018 elastography) shows every MAFLD case has CAP >= 269 dB/m; among participants with BMI >= 25, all with CAP 265-268 are non-MAFLD and all with CAP 269-272 are MAFLD, so steatosis was CAP > 268 dB/m; partial exams (LUAXSTAT = 2) were kept and no IQR/median filter was applied. (5) Refit (R glm, binomial, unweighted, no covariates, Wald CIs): MAFLD on log10(Mn) among rows with log10 Mn < 1.10 (n = 7,131, 3,624 MAFLD) gives OR 3.936 (2.631-5.887), and among rows with log10 Mn >= 1.10 (n = 1,411, 746 MAFLD) OR 0.458 (0.121-1.726), p = 0.2485, matching Table 3; a continuous hinge model (shared intercept) gives 3.546 (2.464-5.105) instead. The Se rows reproduce the same way (36.438, 10.744-123.573; 5.845, 0.893-38.234), and an unadjusted logistic model with Cr, Se, Mn, Cd, Hg and Pb reproduces Figure 3A's Mn 1.028 (1.015-1.040). (6) The paper calls the whole-blood metals 'serum' in places. (7) The CSV's wtint2yr is missing for the pre-pandemic rows, and marital status is missing for them too (Table 1 marital counts cover only 3,264 people).",
  "adjudication": null
 },
 {
  "id": "row191",
  "rank": 60,
  "row": 191,
  "doi": "10.1186/s12888-023-04935-1",
  "pmcid": "PMC10283330",
  "title": "The association between serum albumin and depressive symptoms: a cross-sectional study of NHANES data during 2005–2018",
  "authors": [
   "Zhang, Guimei",
   "Li, Shuna",
   "Wang, Sisi",
   "Deng, Fangyi",
   "Sun, Xizhe",
   "Pan, Jiyang"
  ],
  "year": 2023,
  "journal": "BMC Psychiatry",
  "table_a": {
   "predictor": "Serum albumin",
   "condition": "Depression",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "Compared with the lowest albumin quartile, the multivariate-adjusted effect size (95% confidence interval) for depressive symptoms of the fully adjusted model in the highest albumin quartile was 0.77 (0.60 to 0.99) and − 0.38 (− 0.66 to − 0.09) using logistics regression and linear regression models respectively.",
   "table_location": "Fig. 2 (Weighted association between albumin and depressive symptoms based on logistics regression models; a table with forest plot), row 'Q4', Model III columns 'OR (95% CI)' and 'P'",
   "table_quote": "Fig. 2 (image), row Q4, columns N, Model I OR (95% CI), P, Model II OR (95% CI), P, Model III OR (95% CI), P: \"Q4 | 289 | 0.52 (0.42 to 0.65) | <0.001 | 0.60 (0.48 to 0.76) | <0.001 | 0.77 (0.60 to 0.97) | 0.044\". The Results text agrees with the figure: \"highest albumin quartile (OR = 0.77, 95% CI = 0.60–0.97)\". The abstract gives the upper bound as 0.99 (see notes).",
   "measure": "OR",
   "estimate": 0.77,
   "ci_low": 0.6,
   "ci_high": 0.99,
   "p_value": "0.044 (Fig. 2, Model III, Q4; the abstract prints no p-value)",
   "exposure_contrast": "highest vs lowest serum albumin quartile (Q4 vs Q1, Q1 = reference); quartile cutpoints not reported",
   "model_label": "Model III (fully adjusted), Fig. 2",
   "covariates_in_this_model": [
    "age (years)",
    "race",
    "gender",
    "education level",
    "BMI status",
    "drinking status",
    "smoking status",
    "congestive heart failure",
    "coronary heart disease",
    "liver condition",
    "cancer or malignancy",
    "diabetes",
    "thyroid problem"
   ],
   "n_analytic": 13681,
   "n_quote": "Abstract: \"This cross-sectional study included 13,681 participants aged ≥ 20 years from the NHANES performed during 2005–2018, which produced nationally representative database.\"",
   "events": 1551
  },
  "cycles": [
   "2005-2006",
   "2007-2008",
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=20",
   "inclusion": "NHANES 2005-2018 participants aged 20 or over with complete PHQ-9, serum albumin and covariate data",
   "exclusions": [
    "Missing PHQ-9, albumin data (n=13,586), from n=70,190 to n=56,604 (Fig. 1)",
    "Missing Drinking status, Smoking status, Congestive heart failure, coronary heart disease, Liver condition, Cancer or Malignancy, Diabetes, Thyroid problem data (n=42,923), to n=13,681 (Fig. 1)"
   ],
   "quote": "Methods, Study design: \"The rigorous screening of 70,190 participants identified 13,681 participants aged ≥ 20 years with a complete set of Patient Health Questionnaire-9 (PHQ-9), albumin, and covariate data, who were included in this study.\" Fig. 1 (image): \"Iinitial survey population (2005-2018, age>=20years) n=70,190 | Exclusion: Missing PHQ-9, albumin data (n=13,586) | Met inclusion criteria n=56,604 | Exclusion: Missing Drinking status, Smoking status, Congestive heart failure, coronary heart disease, Liver condition, Cancer or Malignancy, Diabetes, Thyroid problem data (n=42,923) | Enrolled analysis n=13,681\""
  },
  "exposure": {
   "definition": "Serum albumin concentration (refrigerated serum, bromocresol purple dye method, DxC800), divided into quartiles; Q1 (lowest) is the reference. Units for the continuous model are not stated.",
   "nhanes_variables": [
    "LBXSAL (g/dL)",
    "LBDSALSI (g/L)"
   ],
   "nhanes_files": [
    "BIOPRO_D",
    "BIOPRO_E",
    "BIOPRO_F",
    "BIOPRO_G",
    "BIOPRO_H",
    "BIOPRO_I",
    "BIOPRO_J",
    "BIOPRO_L (2021-2023)"
   ],
   "transform": "quartiles (headline Q4 vs Q1); continuous in secondary models",
   "categories": "Quartiles; cutpoints are not reported anywhere (text, Table 1, Figs. 2-4, Supplementary Fig. 1). Table 1 gives quartile sizes that conflict with its own rows: \"| N <colspan=2> | 2745.00 | 2744.00 | 3292.00 | 4900.00 | 13681.00 |\" against row totals of 3971, 3159, 3167 and 3384 (e.g. \"| Gender (%) | Male | 1831 (40.67) | 1788 (51.60) | 2084 (61.17) | 2523 (71.22) | 8226 (56.74) |\" with \"| Female | 2140 (59.33) | 1371 (48.40) | 1083 (38.83) | 861 (28.78) | 5455 (43.26) |\").",
   "quote": "Abstract: \"Serum albumin concentration was measured using the bromocresol purple dye method, and participants were divided into quartiles of serum albumin concentrations.\" Methods, Albumin: \"The DcX800 method is used to measure the albumin concentration as a bichromatic digital endpoint method.\""
  },
  "outcome": {
   "definition": "PHQ-9 total score (nine items, each 0-3, total 0-27); depressive symptoms = total >= 10.",
   "nhanes_variables": [
    "DPQ010",
    "DPQ020",
    "DPQ030",
    "DPQ040",
    "DPQ050",
    "DPQ060",
    "DPQ070",
    "DPQ080",
    "DPQ090"
   ],
   "nhanes_files": [
    "DPQ_D",
    "DPQ_E",
    "DPQ_F",
    "DPQ_G",
    "DPQ_H",
    "DPQ_I",
    "DPQ_J",
    "DPQ_L (2021-2023)"
   ],
   "quote": "Methods, Depressive symptoms: \"Each question is scored from 0 (not at all) to 3 (nearly every day), with the final questionnaire score ranging from 0 to 27. Scores of at least 10 were considered to indicate depressive symptoms [17].\""
  },
  "covariates": [
   {
    "name": "age (years)",
    "coding": "continuous",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "race",
    "coding": "Mexican American, Other Hispanic, Non-Hispanic White, Non-Hispanic Black, Other Race (Table 1)",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "gender",
    "coding": "male, female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "education level",
    "coding": "Less than 9th grade; 9-11th grade; High school graduate/GED or equivalent; Some college or AA degree; College graduate or above (Table 1)",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "BMI status",
    "coding": "continuous kg/m2 (Table 1 reports mean ± SD)",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "how BMI entered the model is not stated"
   },
   {
    "name": "drinking status",
    "coding": "drinking vs nondrinking, by the question 'Do you drink alcohol now?'",
    "nhanes_variables": [
     "unstated (no NHANES item reads 'Do you drink alcohol now?')"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "Current drinking can be built from ALQ_L (ALQ111, ALQ121), but the paper's item is unknown; only 25.23% are 'Yes' in Table 1, far below the share of US adults who drink"
   },
   {
    "name": "smoking status",
    "coding": "smoking vs nonsmoking, by the question 'Do you smoke now?'",
    "nhanes_variables": [
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "SMQ040 is asked only of those with SMQ020 = yes (100+ cigarettes); see notes on a possible ever-smoker restriction"
   },
   {
    "name": "congestive heart failure",
    "coding": "self-reported doctor diagnosis, yes/no",
    "nhanes_variables": [
     "MCQ160B"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "coronary heart disease",
    "coding": "self-reported doctor diagnosis, yes/no",
    "nhanes_variables": [
     "MCQ160C"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "liver condition",
    "coding": "self-reported doctor diagnosis, yes/no ('liver function' in Methods)",
    "nhanes_variables": [
     "MCQ160L"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "cancer or malignancy",
    "coding": "self-reported doctor diagnosis, yes/no",
    "nhanes_variables": [
     "MCQ220"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "diabetes",
    "coding": "yes, no, borderline (Table 1)",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "thyroid problem",
    "coding": "self-reported doctor diagnosis, yes/no",
    "nhanes_variables": [
     "MCQ160M"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   }
  ],
  "design": {
   "weights": "MEC examination weights combined across the seven cycles (WTMEC2YR; the divisor, presumably 7, is not stated)",
   "strata_psu": "not mentioned",
   "quote": "Methods, Statistical analysis: \"All the interviews and MEC examination weights covered in this study are available in the demographic files. The detailed data are available at the following website: https://wwwn.cdc.gov/nchs/nhanes/tutorials/module3.aspx. Since those examined by the MEC were a subset of those interviewed in the survey, we combined the MEC examination weights for the analysis. The NHANES performed during 2005–2018 involved seven survey cycles spanning 14 years, and the data were weighted according to the information that the NCHS analysts provided on how to combine multiple cycles and construct appropriate weights.\"",
   "software": "R 3.6.1",
   "missing_data": "complete case",
   "quote_missing": "Methods, Study design: \"13,681 participants aged ≥ 20 years with a complete set of Patient Health Questionnaire-9 (PHQ-9), albumin, and covariate data, who were included in this study.\""
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "Methods, Statistical analysis: \"Multivariate logistic regression models were used to describe the association between albumin concentration and depressive symptoms. We constructed three models respectively: (1) model I included no adjustment; (2) model II adjusted for age, gender, and race; and (3) model III adjusted for age, race, gender, education level, BMI, drinking status, smoking status, congestive heart failure, coronary heart disease, liver function, cancer or malignancy, diabetes, and thyroid problems.\" and \"All analyses were performed using the statistical software R (version 3.6.1, https://www.r-project.org/).\""
  },
  "unstated": [
   "Albumin quartile cutpoints, and whether quartiles were formed with or without weights and in g/L or g/dL",
   "The albumin unit behind the continuous OR",
   "The NHANES item behind 'drinking status' ('Do you drink alcohol now?' is not an NHANES question)",
   "Whether participants never asked SMQ040 (never smokers) were excluded as missing or coded as nonsmokers",
   "The weight divisor across seven cycles and whether strata and PSUs were used",
   "How BMI entered the model",
   "Handling of PHQ-9 items coded refused/don't know or partly missing",
   "Whether pregnant women were excluded",
   "Reference categories of the categorical covariates"
  ],
  "notes": "DISCREPANCY in the headline CI: the abstract prints 0.77 (0.60 to 0.99); the Results text (\"OR = 0.77, 95% CI = 0.60–0.97\") and Fig. 2 (\"0.77 (0.60 to 0.97)\", P = 0.044) give 0.97. The headline fields hold the abstract's numbers per the headline rule; the software-drawn Fig. 2 is probably the computed value. Other inconsistencies: the abstract's linear Q4 estimate -0.38 (-0.66 to -0.09) differs from Fig. 3 (-0.39, -0.67 to -0.11); the Results text gives the fully adjusted continuous OR as 0.95-0.99 where Fig. 2 has 0.95 to 1.00 (P 0.037), and gives Model III p trend < 0.001 where Fig. 2 has 0.047; Table 1's quartile N row (2745, 2744, 3292, 4900) conflicts with its own category counts (3971, 3159, 3167, 3384). Fig. 1's first box is labeled age >= 20 but its n = 70,190 equals all participants of all ages in 2005-2018 (DEMO_D to DEMO_J documentation counts: 10,348 + 10,149 + 10,537 + 9,756 + 10,175 + 9,971 + 9,254), while those aged 20+ number 39,749 (respondents to the adults-only education item DMDEDUC2 in the same documentation), so 'Met inclusion criteria n=56,604' cannot be adults with PHQ-9 and albumin. Possible ever-smoker restriction (inference, not stated): current smoking comes from 'Do you smoke now?' (SMQ040), which NHANES asks only of people who smoked 100+ cigarettes; dropping participants missing it would remove never smokers. That fits the 42,923 excluded for missing covariates, the 56.74% male share and the 44.52% 'Yes' for current smoking in Table 1 (\"| Smoking status(%) | Yes | 1797 (44.72) | 1450 (44.53) | 1411 (43.86) | 1605 (44.90) | 6263 (44.52) |\"). Drinking 'Yes' is only 25.23% (\"| Drinking status (%) | Yes | 1022 (24.66) | 785 (23.81) | 866 (26.15) | 915 (26.08) | 3588 (25.23) |\"). Methods say 'liver function' but the figures say 'liver condition' (self-reported). In Fig. 4 (subgroups) the Yes/No labels of the chronic conditions look reversed (e.g. congestive heart failure 'Yes' has the narrowest CI although 3.05% have it). Albumin in 2021-2023 is still bromcresol purple (BIOPRO_L documentation: Roche Cobas 8000).",
  "adjudication": null
 },
 {
  "id": "row296",
  "rank": 62,
  "row": 296,
  "doi": "10.1155/2024/4306797",
  "pmcid": "PMC11368549",
  "title": "Associations between Waist Circumference and Sex Steroid Hormones in US Adult Men: Cross-Sectional Findings from the NHANES 2013–2016",
  "authors": [
   "Zhu, Zhisheng",
   "Lin, Xingong",
   "Wang, Chaoyang",
   "Zhu, Shize",
   "Zhou, Xianying"
  ],
  "year": 2024,
  "journal": "International Journal of Endocrinology",
  "table_a": {
   "predictor": "Waist circumference",
   "condition": "Sex steroid hormones",
   "population": "US adult males"
  },
  "headline": {
   "abstract_quote": "After adjusting for confounders, WC was found to be negatively associated with testosterone (β = −0.117, P < 0.001) but positively correlated with estradiol (β = 0.002, P=0.002), especially beyond a WC of 104.5 cm (β = 0.004, P < 0.001).",
   "table_location": "Table 2, column 'Fully adjusted model β, 95% CI, P', row 'Exposure to testosterone' (waist circumference as a continuous variable)",
   "table_quote": "| Exposure to testosterone | −0.117 (−0.127, −0.107) <0.001 | −0.120 (−0.131, −0.108) <0.001 | −0.117 (−0.136, −0.098) <0.001 |",
   "measure": "beta",
   "estimate": -0.117,
   "ci_low": -0.136,
   "ci_high": -0.098,
   "p_value": "<0.001",
   "exposure_contrast": "per 1 cm increase in waist circumference (continuous); outcome is the square root of serum total testosterone (3.4: 'For every 1 cm increase in WC, the square root of testosterone is reduced by 0.117 units')",
   "model_label": "Fully adjusted model",
   "covariates_in_this_model": [
    "age",
    "race",
    "education level",
    "poverty income ratio",
    "diabetes",
    "session of blood sample collection",
    "cotinine",
    "alcohol intake",
    "smoking status",
    "physical activity"
   ],
   "n_analytic": 3359,
   "n_quote": "For our study, we included 3,359 adult men over the age of 20 years with data on WC, serum total testosterone (TT), estradiol (E2), and sex hormone-binding globulin (SHBG) from the 2013–2016 NHANES. (2.1 Study Design and Participants); Table 1: '| Number | 3359 | 3359 | 836 | 843 | 840 | 840 |  |'",
   "events": null
  },
  "cycles": [
   "2013-2014",
   "2015-2016"
  ],
  "population": {
   "age": ">=20",
   "inclusion": "Men aged 20 and older in NHANES 2013-2016 with waist circumference, total testosterone, estradiol, SHBG, BMI, and alcohol data, after excluding hormone-affecting cancer, thyroid disorders, liver disease, glucocorticoid use, recent infection, and HIV/AIDS.",
   "exclusions": [
    "From 'NHANES 2013-2016 (n=20146)' to 'Adult male participants aged 20 years and older (n=5290)' (Figure 1)",
    "'Missing covariates (n=1219): sex hormones (n=751), WC (n=210), BMI (n=14), alcohol consumption (n=244), education, race, gender, age, and smoking (n=0 for each).' Remaining n=4071 (Figure 1)",
    "'Cancer affecting hormones (n=133), thyroid disorders (n=171), liver diseases (n=211, including hepatitis B and hepatitis C n=46)'. Remaining n=3556 (Figure 1)",
    "'Glucocorticoid use (n=43), recent infection (n=137), HIV/AIDS (n=17)'. Final n=3359, printed 'Participants in fianl analyses (n=3359)' (Figure 1)"
   ],
   "quote": "For our study, we included 3,359 adult men over the age of 20 years with data on WC, serum total testosterone (TT), estradiol (E2), and sex hormone-binding globulin (SHBG) from the 2013–2016 NHANES. Our selection process is shown in Figure 1. (2.1) ... Figure 1 (transcribed from the flowchart image): NHANES 2013-2016 (n=20146) | Adult male participants aged 20 years and older (n=5290) | Missing covariates (n=1219): sex hormones (n=751), WC (n=210), BMI (n=14), alcohol consumption (n=244), education, race, gender, age, and smoking (n=0 for each). | Remaining participants (n=4071) | Cancer affecting hormones (n=133), thyroid disorders (n=171), liver diseases (n=211, including hepatitis B and hepatitis C n=46) | Remaining participants (n=3556) | Glucocorticoid use (n=43), recent infection (n=137), HIV/AIDS (n=17) | Participants in fianl analyses (n=3359)"
  },
  "exposure": {
   "definition": "Waist circumference (cm), measured in the MEC by trained staff at a horizontal line over the uppermost lateral edge of the right iliac crest; continuous, per 1 cm (quartiles are used only in the categorical models).",
   "nhanes_variables": [
    "BMXWAIST"
   ],
   "nhanes_files": [
    "BMX"
   ],
   "transform": "none (continuous, per 1 cm)",
   "categories": "Not used for the headline. Quartiles in Tables 1 and 2: 'WC Q1 (65.80 cm–89.70 cm) | WC Q2 (89.80 cm–99.40 cm) | WC Q3 (99.50 cm −109.30 cm) | WC Q4 (109.40 cm −162.70 cm)' (Table 1 header)",
   "quote": "To measure WC, a horizontal line was drawn over the uppermost external edge of the right iliac bone. (2.3) ... For every 1 cm increase in WC, the square root of testosterone is reduced by 0.117 units (β = −0.117, P < 0.001). (3.4)"
  },
  "outcome": {
   "definition": "Serum total testosterone (ID-LC-MS/MS), square-root transformed. The paper does not say which unit was square-rooted; Table 2's coefficients fit ng/dL (see unstated and notes).",
   "nhanes_variables": [
    "LBXTST"
   ],
   "nhanes_files": [
    "TST"
   ],
   "quote": "The concentrations of TT and E2 were measured using isotope dilution high-performance liquid chromatography-tandem mass spectrometry (ID-LC-MS/MS). (2.2) ... To indirectly estimate the levels of circulating free androgens and aromatase activity, we standardized the units of total testosterone and estradiol to nanomoles per liter (nmol/L), ensuring accuracy in our calculations. (2.2) ... Due to the right-skewed distribution of sex hormone indicators, we applied a log2 transformation to SHBG, while testosterone, estradiol, FAI, and T/E2 underwent square root transformation. (2.5)"
  },
  "covariates": [
   {
    "name": "age",
    "coding": "continuous (years)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "race",
    "coding": "Mexican American, other Hispanic, non-Hispanic White, non-Hispanic Black, non-Hispanic Asian, or other race",
    "nhanes_variables": [
     "RIDRETH3"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Reference level unstated (Table 1 lists Mexican American first)."
   },
   {
    "name": "education level",
    "coding": "less than 9th grade, 9−11th grade, high school graduate/GED or equivalent, some college or AA degree, or college graduate or above",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Figure 1 reports no exclusions for missing education."
   },
   {
    "name": "poverty income ratio",
    "coding": "continuous",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "How missing PIR was handled is unstated (not among Figure 1's exclusions)."
   },
   {
    "name": "diabetes",
    "coding": "categorical (yes/no): HbA1c >= 6.5% or fasting plasma glucose >= 126 mg/dL, or yes to 'Take diabetic pills to lower blood sugar?', 'Doctor told you have diabetes?', or 'Taking insulin now?'",
    "nhanes_variables": [
     "LBXGH",
     "LBXGLU",
     "DIQ070",
     "DIQ010",
     "DIQ050"
    ],
    "nhanes_files": [
     "GHB",
     "GLU",
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "All components are in 2021-2023 (GHB_L, GLU_L fasting subsample, DIQ_L). Treatment of DIQ010 'borderline' is unstated."
   },
   {
    "name": "session of blood sample collection",
    "coding": "morning, afternoon, or evening",
    "nhanes_variables": [
     "PHDSESN"
    ],
    "nhanes_files": [
     "FASTQX"
    ],
    "in_2021_2023": false,
    "note": "2021-2023 FASTQX_L has PHDSESNZ coded only 0 = morning, 1 = afternoon/evening (FASTQX_L codebook), so the three-level coding cannot be rebuilt; a two-level version can."
   },
   {
    "name": "cotinine",
    "coding": "continuous (serum cotinine)",
    "nhanes_variables": [
     "LBXCOT"
    ],
    "nhanes_files": [
     "COT"
    ],
    "in_2021_2023": true,
    "note": "Any transformation is unstated; Table 1 reports median (Q1-Q3)."
   },
   {
    "name": "alcohol intake",
    "coding": "1–5 drinks/month, 5–10 drinks/month, 10+drinks/month, or nondrinker (a drink is a 12 oz beer, a 5 oz glass of wine, or 1.5 oz of liquor)",
    "nhanes_variables": [
     "ALQ101",
     "ALQ110",
     "ALQ120Q",
     "ALQ120U",
     "ALQ130"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "The paper names no items; these are the likely 2013-2016 ones. How drinks/month and nondrinkers were derived, and the overlapping boundaries at 5 and 10, are unstated. 2021-2023 ALQ_L has ALQ111 (ever drank), ALQ121 (12-month frequency, categorical) and ALQ130 (drinks per drinking day), so drinks/month can be approximated. Figure 1 excluded 244 men missing alcohol."
   },
   {
    "name": "smoking status",
    "coding": "never smoker, former smoker, or current smoker ('Participants who currently smoke or have smoked more than 100 cigarettes in their lifetime are identified as current or former smokers, respectively')",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "Both items are in SMQ_L."
   },
   {
    "name": "physical activity",
    "coding": "categorical; Table 1: 'Physical activity time (hour/week)' Nonactivity, 0.1–0.9, 1.0–3.4, 3.5–5.9, ≥6",
    "nhanes_variables": [],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": false,
    "note": "Which domains and intensities were summed into hours/week is unstated. 2021-2023 PAQ_L has only leisure-time moderate and vigorous activity (PAD790Q/U, PAD800, PAD810Q/U, PAD820), not the work and transport items of the 2013-2016 questionnaire, so it can be rebuilt only if the paper used leisure-time activity."
   },
   {
    "name": "BMI",
    "coding": "continuous; listed as a covariate but omitted from all regression models",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Not in the fully adjusted model ('we omitted the BMI as a confounder'); Figure 1 excluded 14 men missing BMI."
   }
  ],
  "design": {
   "weights": "sampling weights; which weight variable and how the two cycles were combined are unstated",
   "strata_psu": "unstated",
   "quote": "We used sampling weights to adjust for selection probabilities, oversampling, nonresponse, and demographic discrepancies between the sample and the entire US population. (2.5) ... All models were weighted. (Table 2 note)",
   "software": "R software (version 4.2.2) and EmpowerStats",
   "missing_data": "complete case for sex hormones, WC, BMI, and alcohol (Figure 1 exclusions); how missing PIR, cotinine, diabetes inputs, blood-draw session, and physical activity were handled is unstated",
   "quote_missing": "Figure 1 (transcribed): 'Missing covariates (n=1219): sex hormones (n=751), WC (n=210), BMI (n=14), alcohol consumption (n=244), education, race, gender, age, and smoking (n=0 for each).'"
  },
  "model": {
   "family": "linear",
   "weighted": true,
   "quote": "We used weighted multivariable linear regression models to examine the relationship between WC and sex hormones, including adjustments for various confounding factors such as age, race, education level, poverty income ratio, diabetes, session of blood sample collection, cotinine, alcohol intake, smoking status, and physical activity. (2.5) ... We analyzed the data using R software (version 4.2.2) and EmpowerStats (https://www.empowerstats.com). (2.5)"
  },
  "unstated": [
   "Which survey weight was used ('sampling weights' only) and how the two 2-year cycles were combined",
   "Whether strata and PSU (SDMVSTRA, SDMVPSU) were used for variance estimation",
   "Units of testosterone before the square-root transform: TT was converted to nmol/L for FAI and T/E2, but Table 2's coefficients fit ng/dL (see notes)",
   "Handling of hormone values below the limit of detection",
   "Which cancers counted as 'cancer affecting hormones'",
   "How thyroid disorders and liver diseases were identified (self-report items, medications, or serology)",
   "How glucocorticoid use, recent infection, and HIV/AIDS were identified",
   "How drinks/month and nondrinkers were derived from the alcohol items, and the boundaries of the overlapping categories (1–5, 5–10, 10+)",
   "Which physical activity domains and intensities were summed into hours/week",
   "How missing PIR, cotinine, diabetes inputs, blood-draw session, and physical activity were handled (Figure 1 lists exclusions only for sex hormones, WC, BMI, and alcohol)",
   "Reference levels of the categorical covariates",
   "Whether DIQ010 'borderline' counted as diabetes"
  ],
  "notes": "Headline choice: Table A's condition covers five outcomes (testosterone, estradiol, SHBG, FAI, T/E2). The abstract's first estimate is for testosterone (β = −0.117) but has no 95% CI, so per the brief the headline is the most-adjusted continuous estimate in the main regression table: Table 2, 'Exposure to testosterone', fully adjusted model, the same estimate with its CI. Eligibility: only the population fails E5, through Figure 1's exclusions of glucocorticoid use (n=43) and HIV/AIDS (n=17), plus recent infection (n=137, method unstated): 197 of 3,556 men. If those exclusions were waived, the rest of the population, the exposure, and the outcome are all constructible in 2021-2023. Sources: Figure 1's counts are not in the text conversion; they were read from the Figure 1 image in the PMC open-access copy (saved to scratchpad/dl/supp/row296/fig1.jpg). Supplementary Tables 1-4 were fetched from the same copy (scratchpad/dl/supp/row296/f1zip/). Testosterone units (computed check, not stated by the paper): unweighted Table 1 quartile means 535.34 (Q1) and 331.86 (Q4) ng/dl give a square-root difference of −4.92 on the ng/dL scale versus −0.917 on the nmol/L scale; Table 2's nonadjusted Q4 vs Q1 coefficient is −4.680, and −0.117 per cm times the 39.28 cm gap between Q1 and Q4 mean WC is −4.60, so the square root was taken of ng/dL. Analysis in R 4.2.2 and EmpowerStats; GAM smoothing and two-piecewise models for nonlinearity; subgroups via stratified models. BMI was listed as a covariate but dropped from all models (Spearman 0.91 with WC). Paper errors: Table 2's T/E2 fully adjusted Q2 prints '−0.932 (−0.150, −0.354) 0.006', a CI that excludes its estimate; Figure 1 prints 'fianl'. For the replication: TST_L carries the phlebotomy weight WTPH2YR; the paper's other hormones are LBXEST and LBXSHBG.",
  "adjudication": "The extraction marked E5 failed because three exclusion steps (glucocorticoid use, HIV/AIDS, recent infection) can't be built in 2021-2023. Under E5 those are exclusion steps, not defining characteristics (the population is men aged 20 and older), so the paper is eligible; the harmonized version leaves the three steps out."
 },
 {
  "id": "row101",
  "rank": 63,
  "row": 101,
  "doi": "10.1186/s40001-022-00977-5",
  "pmcid": "PMC9824928",
  "title": "The association between blood manganese and liver stiffness in participants with chronic obstructive pulmonary disease: a cross-sectional study from NHANES 2017–2018",
  "authors": [
   "Han, Kexing",
   "Shen, Jiapei",
   "Tan, Kexuan",
   "Liu, Jiaying",
   "Sun, Weijie",
   "Gao, Yufeng"
  ],
  "year": 2023,
  "journal": "European Journal of Medical Research",
  "table_a": {
   "predictor": "Blood manganese",
   "condition": "Liver stiffness",
   "population": "US adults with chronic obstructive pulmonary disease"
  },
  "headline": {
   "abstract_quote": "Among the 4690 participants, blood manganese was lower in the COPD group but liver stiffness was higher (p < 0.05). There was a positive correlation between blood manganese and liver stiffness (β = 0.08, 95% CI 0.03, 0.12).",
   "table_location": "Table 3, column 'Model 3', row 'Total'",
   "table_quote": "| Total | 0.11 (0.07, 0.15) | 0.15 (0.11, 0.19) | 0.08 (0.03, 0.12) |",
   "measure": "beta",
   "estimate": 0.08,
   "ci_low": 0.03,
   "ci_high": 0.12,
   "p_value": "not reported (Table 3 prints no P values; the abstract gives none for this estimate)",
   "exposure_contrast": "per 1 ug/L increase in blood manganese (continuous, untransformed; the scaling is implied by the units, not stated)",
   "model_label": "Model 3 (full adjustment model)",
   "covariates_in_this_model": [
    "age",
    "sex",
    "race",
    "BMI",
    "PIR",
    "smoking",
    "alcohol consumption",
    "dietary status",
    "prescription medication use",
    "COPD",
    "diabetes",
    "congestive heart failure",
    "liver condition"
   ],
   "n_analytic": 4690,
   "n_quote": "The final 4690 participants were included in this study. (Methods, Participants)",
   "events": null
  },
  "cycles": [
   "2017-2018"
  ],
  "population": {
   "age": ">=20",
   "inclusion": "NHANES 2017-2018 adults aged 20 and older who were asked 'Ever told you had COPD?' and had liver transient elastography, blood manganese, diet status, and smoking data; COPD and non-COPD participants together.",
   "exclusions": [
    "Of 9254 sampled, those asked the COPD question were limited to ages 20 and older: 5569",
    "no liver transient elastography data (n = 697)",
    "no blood manganese data (n = 173)",
    "no information on diet status (n = 7)",
    "no information on smoking (n = 2)",
    "final n = 4690"
   ],
   "quote": "A total of 9254 participants were sampled for NHANES 2017–2018, of which those who participated in the question \" Ever told you had COPD?\" were limited to no less than 20 years of age. A total of 5569 participants took part in the questionnaire, of which 697 had no liver transient elastography data, 173 had no blood manganese data, 7 had no information on diet status and 2 had no information on smoking. The final 4690 participants were included in this study. (Methods, Participants)"
  },
  "exposure": {
   "definition": "Blood manganese (ug/L) from the laboratory data (NHANES measures it in whole blood; the paper once calls it 'serum manganese'); continuous.",
   "nhanes_variables": [
    "LBXBMN"
   ],
   "nhanes_files": [
    "PBCD"
   ],
   "transform": "none (continuous, per 1 ug/L; implied, not stated)",
   "categories": null,
   "quote": "Liver stiffness (kPa) was measured from “Examination Date” and blood manganese (ug/L) was obtained from “Laboratory Data”. (Abstract) ... Serum manganese (ug/L) data were obtained from laboratory test. (Methods, Variables)"
  },
  "outcome": {
   "definition": "Median liver stiffness (kPa) from liver transient elastography; continuous.",
   "nhanes_variables": [
    "LUXSMED"
   ],
   "nhanes_files": [
    "LUX"
   ],
   "quote": "Median liver stiffness data for the primary outcome indicator were obtained from the examination date, which included complete liver transient elastography information. (Methods, Variables)"
  },
  "covariates": [
   {
    "name": "age",
    "coding": "years; Table 1 gives mean ± SD and groups < 60 / ≥ 60 (form in the model unstated)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "sex",
    "coding": "male, female ('gender')",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "race",
    "coding": "White, Black, Other race (Table 1); Table 4 labels 'White', 'Non-Hispanic Black', 'Other race'",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "The recode (which NHANES groups form 'White' and 'Other race', RIDRETH1 or RIDRETH3) is unstated."
   },
   {
    "name": "BMI",
    "coding": "kg/m2; Table 1 gives mean ± SD and groups < 28 / ≥ 28 (form in the model unstated)",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "PIR",
    "coding": "income to poverty ratio grouped < 1.39, 1.39–3.49, > 3.49, Unclear (Table 1)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Missing PIR forms the 'Unclear' level (12.69% of non-COPD, 11.21% of COPD in Table 1)."
   },
   {
    "name": "smoking",
    "coding": "smoked at least 100 cigarettes in life (yes/no)",
    "nhanes_variables": [
     "SMQ020"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "2 participants were excluded for missing smoking."
   },
   {
    "name": "alcohol consumption",
    "coding": "drinking alcohol more than 2 times per week in the past 12 months (yes, no, unclear)",
    "nhanes_variables": [
     "ALQ121"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "ALQ121 (12-month drinking frequency) is in ALQ_L; which answers count as 'more than 2 times per week' is unstated."
   },
   {
    "name": "dietary status",
    "coding": "answer to 'How healthy is the diet?' (excellent, very good, good, fair, poor)",
    "nhanes_variables": [
     "DBQ700"
    ],
    "nhanes_files": [
     "DBQ"
    ],
    "in_2021_2023": false,
    "note": "DBQ_L (2021-2023) has no DBQ700. 7 participants were excluded for missing diet status."
   },
   {
    "name": "prescription medication use",
    "coding": "taken prescription medicine, past month (yes, no, unclear)",
    "nhanes_variables": [
     "RXDUSE"
    ],
    "nhanes_files": [
     "RXQ_RX"
    ],
    "in_2021_2023": true,
    "note": "The 2021-2023 item is RXQ033 ('Taken prescription medicine, past month') in RXQ_RX_L."
   },
   {
    "name": "COPD",
    "coding": "'Ever been told you had COPD?' (yes/no)",
    "nhanes_variables": [
     "MCQ160o"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": false,
    "note": "2021-2023 has only MCQ160p, which asks about COPD, emphysema, or chronic bronchitis together, a broader construct (2017-2018 asked emphysema MCQ160g and chronic bronchitis MCQ160k separately). Whether COPD entered the 'Total' model is ambiguous (see unstated)."
   },
   {
    "name": "diabetes",
    "coding": "'Doctor told you have diabetes?' (yes, no, borderline, unclear)",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "congestive heart failure",
    "coding": "'Ever been told you had congestive heart failure?' (yes, no, unclear)",
    "nhanes_variables": [
     "MCQ160b"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "MCQ160b is in MCQ_L."
   },
   {
    "name": "liver condition",
    "coding": "'ever been told that you have liver condition?' (yes, no, unclear)",
    "nhanes_variables": [
     "MCQ160l"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "MCQ160l is in MCQ_L."
   }
  ],
  "design": {
   "weights": "2-year sampling weights; which weight variable is unstated",
   "strata_psu": "unstated",
   "quote": "To make the participants more representative, 2-year sampling weights were adopted in the analysis. (Methods, Statistical analysis)",
   "software": "R and EmpowerStats",
   "missing_data": "complete case for the COPD item, elastography, blood manganese, diet status, and smoking; for other covariates, mean imputation for continuous ones with under 12% missing, otherwise grouping with an 'unclear group', and a missing category ('Unclear group') for categorical ones",
   "quote_missing": "Continuous variables were expressed as mean ± standard deviation, and when the missing sample size was less than 12%, the mean was used instead. Otherwise, continuous variables were set as \"unclear group\" after grouping. Categorical variables were represented as percentages, and missing data were defined as 'Unclear group'. (Methods, Statistical analysis)"
  },
  "model": {
   "family": "linear",
   "weighted": true,
   "quote": "Association between blood manganese and liver stiffness was analyzed by multiple linear regression model. According to different adjustment variables, Model 1: no adjustment model, Model 2: adjustment of age, gender and race, and Model 3: full adjustment model are generated. (Methods, Statistical analysis) ... All data collection and analysis were performed through R (http://www.R-project.org) and EmpowerStats (http://www.enpowerstats.com), and P < 0.05 was considered statistically significant."
  },
  "unstated": [
   "Which 2-year weight was used (e.g. WTMEC2YR) and whether strata and PSU were used",
   "Whether COPD entered the 'Total' Model 3: the Methods list COPD as a covariate, but Table 3's note says 'all the covariates in Table 1', and Table 1 is split by COPD",
   "Whether age and BMI entered as continuous or as Table 1's groups (< 60 / ≥ 60 years; < 28 / ≥ 28 kg/m2)",
   "The race recode behind 'White', 'Black', and 'Other race'",
   "Which ALQ121 answers count as 'drinking alcohol more than 2 times per week'",
   "Exact PIR cut boundaries (< 1.39, 1.39–3.49, > 3.49)",
   "Which continuous covariates were mean-imputed",
   "How 'don't know' answers to the COPD, CHF, liver, and diabetes items were coded (the 5569 base includes 9 'don't know' answers to MCQ160o per the codebook)",
   "Elastography quality criteria (exam status, IQR/median ratio, fasting) beyond 'complete liver transient elastography information'",
   "Whether blood manganese outliers were excluded from the Table 3 models (they were excluded for curve fitting, which used 0-35 ug/L)",
   "Reference levels of the categorical covariates",
   "The exposure scale (per 1 ug/L, untransformed) is implied, not stated"
  ],
  "notes": "Headline choice: the abstract's first estimate with a 95% CI is for all 4690 participants (COPD and non-COPD together), so per the brief it is the headline and COPD is a covariate. Table A's population label (adults with COPD) matches the title, but the COPD estimate (Table 3, Model 3, COPD row: 0.25 (0.08, 0.42), n = 223) is a subgroup. If the COPD subgroup were targeted instead: its population rests on the 2017-2018 item MCQ160o ('Ever told you had COPD?', 293 yes of 5,569 in the MCQ_J codebook), while 2021-2023 asks only MCQ160p (COPD, emphysema, or chronic bronchitis together; 2017-2018 asked emphysema MCQ160g, 106 yes, and chronic bronchitis MCQ160k, 395 yes, separately), a broader construct, so that population would arguably fail E5. Exposure wording: the paper says 'Serum manganese' once, but NHANES measures manganese in whole blood (LBXBMN, 'Blood manganese (ug/L)'). Software: R and EmpowerStats (URL printed as 'www.enpowerstats.com'). Outliers: 'To explore the non-linear relationship between blood manganese and liver stiffness, we performed a smooth curve fitting model analysis and excluded significant blood manganese outliers.'; 'After excluding significant outliers we set the blood manganese level to 0-35ug/L (Fig. 3)'. Table 5's 'Linear effect model' Total is 0.08 (0.04, 0.12), against Table 3's 0.08 (0.03, 0.12) (COPD 0.24 (0.06, 0.41) against 0.25 (0.08, 0.42)), perhaps because Table 5 used the 0-35 ug/L sample (unstated). The threshold-model rule is garbled ('choose a non-effect model when the log-likelihood ratio (LLR) was < 0.05'). Table oddities: Table 1 Smoker '| Yes | 14.80 | 60.0 |  |' looks reversed for COPD; Table 2 '| < 120 | 37.67 | 209.46 |  |' gives an impossible 209.46%. For the replication: PBCD_L carries the phlebotomy weight WTPH2YR; LUX_L is a MEC exam.",
  "adjudication": null
 },
 {
  "id": "row068",
  "rank": 64,
  "row": 68,
  "doi": "10.1038/s41598-022-05124-y",
  "pmcid": "PMC8782988",
  "title": "Association between sleep duration on workdays and blood pressure in non-overweight/obese population in NHANES: a public database research",
  "authors": [
   "Su, Yingjie",
   "Li, Changluo",
   "Long, Yong",
   "He, Liudang",
   "Ding, Ning"
  ],
  "year": 2022,
  "journal": "Scientific Reports",
  "table_a": {
   "predictor": "Sleep health",
   "condition": "Blood pressure",
   "population": "US adults with obesity"
  },
  "headline": {
   "abstract_quote": "Compared with sleep duration of 6–8 h, both sleep duration < 6 h and ≥ 8 h on workdays were significantly associated with increased SBP (β, 3.58 [95% CI 1.60, 5.56] and 1.70 [95% CI 0.76, 2.64], respectively).",
   "table_location": "Table 3, column 'Model II (β, 95% CI, P)', row '< 6 h' (reference '6–8 h')",
   "table_quote": "| < 6 h | 6.15 (3.88, 8.42) < 0.0001 | 4.17 (2.19, 6.15) < 0.0001 | 3.58 (1.60, 5.56) 0.0004 |",
   "measure": "beta",
   "estimate": 3.58,
   "ci_low": 1.6,
   "ci_high": 5.56,
   "p_value": "0.0004",
   "exposure_contrast": "sleep duration on workdays < 6 h vs 6–8 h (reference); SBP in mmHg",
   "model_label": "Model II",
   "covariates_in_this_model": [
    "Gender",
    "Age",
    "Race",
    "alcohol",
    "Albumin",
    "Creatinine",
    "Hemoglobin",
    "diabetes",
    "hypertension",
    "snort or stop breathing",
    "smoke",
    "TC",
    "BMI",
    "AST",
    "HDL"
   ],
   "n_analytic": 2887,
   "n_quote": "Finally, 2887 participants were included in the study (Fig. 1). (Methods, Study population); Table 1 header: '|  | Total(n = 2887) | Male(n = 1378) | Female(n = 1509) | P value |'",
   "events": null
  },
  "cycles": [
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": "unstated; the counts imply everyone aged 16 and older (see notes)",
   "inclusion": "NHANES 2015-2018 participants with workday sleep duration and blood pressure, not taking antihypertensive medication, with BMI < 25 kg/m2 (underweight included).",
   "exclusions": [
    "Start: 19,225 participants in 2015-2018 NHANES",
    "missing sleep duration data (n = 6818); 12,407 remain (Figure 1)",
    "missing BP data (n = 1055); 11,352 remain (Figure 1)",
    "taking antihypertensive medications (n = 2944); 8408 remain (Figure 1)",
    "missing BMI and BMI ≥ 25 (n = 5521); 2887 remain (Figure 1)"
   ],
   "quote": "According to WHO guidelines, BMI is divided into underweight (< 18.5 kg/m2), normal weight (18.5–24.99 kg/m2), overweight (25–29.99 kg/m2) and obesity (≥ 30 kg/m2)19. Non-overweight/obese is defined as people with BMI < 25. This research combined 2015–2018 data for analysis. A total of 19,225 potentially participants were enrolled, 16,338 participants were excluded for the following reasons: missing sleep duration data(n = 6818), missing BP data(n = 1055), taking antihypertensive medications(n = 2944), missing BMI and BMI ≥ 25(n = 5521). Finally, 2887 participants were included in the study (Fig. 1). (Methods, Study population)"
  },
  "exposure": {
   "definition": "Self-reported usual hours of sleep on weekdays or workdays, in three groups: < 6 h, 6–8 h (reference), ≥ 8 h.",
   "nhanes_variables": [
    "SLD012"
   ],
   "nhanes_files": [
    "SLQ"
   ],
   "transform": "categories: < 6 h, 6–8 h (reference), ≥ 8 h",
   "categories": "'Sleep duration was divided into three groups, which were < 6 h, 6–8 h, ≥ 8 h respectively, of which 6–8 h was used as the reference group.' (Methods, Definition). The labels imply 6 h belongs to 6–8 h and 8 h to ≥ 8 h.",
   "quote": "Sleep duration on workdays was evaluated by the questionnaire with the following questions: “Number of hours usually sleep on weekdays or workdays”. Sleep duration was divided into three groups, which were < 6 h, 6–8 h, ≥ 8 h respectively, of which 6–8 h was used as the reference group. (Methods, Definition)"
  },
  "outcome": {
   "definition": "Systolic blood pressure (mmHg): average of all available auscultatory readings (three consecutive readings, plus a fourth if one was not completed).",
   "nhanes_variables": [
    "BPXSY1",
    "BPXSY2",
    "BPXSY3",
    "BPXSY4"
   ],
   "nhanes_files": [
    "BPX"
   ],
   "quote": "The trained and certified examiners used the standardized protocols and calibrated equipment to get the blood pressure readings. Three consecutive BP readings were acquired via ausculatory means. If a BP measurement was not successfully completed, a fourth measurement was implemented. The average of all available measures was used. (Methods, Definition)"
  },
  "covariates": [
   {
    "name": "Gender",
    "coding": "categorical (male, female)",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "Age",
    "coding": "continuous (years)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "Race",
    "coding": "Mexican American, white, black, other race",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Which groups form 'other race' (presumably other Hispanic and other/multiracial) is unstated; Table 2 uses Mexican American as reference."
   },
   {
    "name": "alcohol",
    "coding": "response to 'In the past 12 months, how often did you drink any type of alcoholic beverage?': drinking, no drinking, not recorded",
    "nhanes_variables": [
     "ALQ120Q",
     "ALQ121"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "2015-2016 asked ALQ120Q (with ALQ120U); 2017-2018 and 2021-2023 ask ALQ121. Alcohol items target ages 18+, so 16-17-year-olds and skip-pattern nonrespondents fall in 'not recorded' (24.77% in Table 1)."
   },
   {
    "name": "Albumin",
    "coding": "continuous (g/L)",
    "nhanes_variables": [
     "LBDSALSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Missing (7.8%) handled by a dummy indicator."
   },
   {
    "name": "Creatinine",
    "coding": "continuous (umol/L)",
    "nhanes_variables": [
     "LBDSCRSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Missing (7.8%) handled by a dummy indicator."
   },
   {
    "name": "Hemoglobin",
    "coding": "continuous (g/dL)",
    "nhanes_variables": [
     "LBXHGB"
    ],
    "nhanes_files": [
     "CBC"
    ],
    "in_2021_2023": true,
    "note": "Missing (6.3%) handled by a dummy indicator."
   },
   {
    "name": "diabetes",
    "coding": "response to 'Have you ever been told by a doctor or health professional that you have diabetes or sugar diabetes?': yes, no, borderline, not recorded",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "hypertension",
    "coding": "response to 'Have you ever been told by a doctor or other health professional that you had hypertension, also called high blood pressure?': yes, no, not recorded",
    "nhanes_variables": [
     "BPQ020"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Table 1 shows 50.04% 'Not recorded', although BPQ020 had no missing values among ages 16+ in the 2015-2016 and 2017-2018 codebooks; the coding is unclear."
   },
   {
    "name": "snort or stop breathing",
    "coding": "response to 'In the past 12 months, how often did you snort, gasp, or stop breathing while you were asleep?': yes, no, not recorded",
    "nhanes_variables": [
     "SLQ040"
    ],
    "nhanes_files": [
     "SLQ"
    ],
    "in_2021_2023": false,
    "note": "SLQ_L (2021-2023) has only sleep and wake times and sleep hours; no SLQ040. How the frequency answers (never, rarely, occasionally, frequently, don't know) became yes/no is unstated."
   },
   {
    "name": "smoke",
    "coding": "response to 'Do you now smoke cigarettes?': smoking, not smoking, not recorded",
    "nhanes_variables": [
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "SMQ040 is asked only of people who smoked 100+ cigarettes (SMQ020), so never smokers likely sit in 'not recorded' (66.53% in Table 1)."
   },
   {
    "name": "TC",
    "coding": "continuous (mmol/L)",
    "nhanes_variables": [
     "LBDTCSI"
    ],
    "nhanes_files": [
     "TCHOL"
    ],
    "in_2021_2023": true,
    "note": "Missing (7.7%) handled by a dummy indicator."
   },
   {
    "name": "BMI",
    "coding": "continuous (kg/m2)",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "AST",
    "coding": "continuous (IU/L)",
    "nhanes_variables": [
     "LBXSASSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Missing (8.0%) handled by a dummy indicator."
   },
   {
    "name": "HDL",
    "coding": "continuous (mmol/L)",
    "nhanes_variables": [
     "LBDHDDSI"
    ],
    "nhanes_files": [
     "HDL"
    ],
    "in_2021_2023": true,
    "note": "Missing (7.7%) handled by a dummy indicator."
   }
  ],
  "design": {
   "weights": "NHANES sample weights; which weight and how the two cycles were combined are unstated",
   "strata_psu": "unstated",
   "quote": "All estimates were calculated accounting for NHANES sample weights. (Methods, Statistical analysis)",
   "software": "R and EmpowerStats",
   "missing_data": "missing indicator: missing continuous covariates (albumin, hemoglobin, creatinine, TC, AST, HDL) flagged by dummy variables; missing categorical covariates kept as their own group ('not recorded')",
   "quote_missing": "The values of missing continuous covariates were indicated by dummy variables, including albumin, hemoglobin, creatinine and TC, AST, HDL and the missing ratios were 7.8%, 6.3%, 7.8%, 7.7%, 8.0%, 7.7% respectively. The missing categorical variables were included in the analysis as a single group. (Methods, Statistical analysis)"
  },
  "model": {
   "family": "linear",
   "weighted": true,
   "quote": "A weighted multiple linear regression model was used to assess the correlation between sleep duration on workdays and BP including systolic blood pressure(SBP) and diastolic blood pressure(DBP). The covariates mentioned above were adjusted as potential effect modifiers. (Methods, Statistical analysis) ... The statistical software packages R (http://www.R-project.org) and EmpowerStats (http://www.empowerstats.com) were used for the data analyses."
  },
  "unstated": [
   "Which survey weight (e.g. WTMEC2YR) and how the two cycles were combined; whether strata and PSU were used",
   "Age range: no age limit is stated (see notes)",
   "Boundary handling for exactly 6 h and 8 h (the labels '< 6 h' and '≥ 8 h' imply 6 to < 8 h as the reference)",
   "Which item identified 'taking antihypertensive medications' (BPQ050A is the likely one) and how missing or don't-know answers were treated",
   "What counted as 'missing BP data' (no reading at all, or fewer than some number), and whether zero readings were dropped",
   "How the alcohol item was harmonized across cycles (ALQ120Q in 2015-2016, ALQ121 in 2017-2018) and what 'not recorded' covers",
   "How SLQ040's frequency answers were split into yes, no, and not recorded",
   "How the hypertension covariate was coded (50.04% 'Not recorded' in Table 1)",
   "Which RIDRETH1 groups form 'other race'",
   "Reference levels of the categorical covariates in Model II (Table 2 uses male, Mexican American, no drinking, diabetes yes, smoking, hypertension yes, and snort no)",
   "Whether the missing-indicator method set the missing continuous values to a constant (e.g. 0 or the mean)"
  ],
  "notes": "Table A's population label ('US adults with obesity') is wrong: the paper studies the non-overweight/obese population ('Non-overweight/obese is defined as people with BMI < 25.'), BMI < 25 kg/m2, underweight included. Table A's predictor 'Sleep health' is workday sleep duration only (SLD012). Ages: the paper states no age limit. The 12,407 left after the sleep exclusion equal the non-missing SLD012 counts in the 2015-2016 SLQ_I (6,294) and 2017-2018 SLQ_J (6,113) codebooks, whose target is ages 16+, so the sample is everyone aged 16+ with sleep hours, including 16-19-year-olds; 2021-2023 SLQ_L also targets ages 16+. Exposure item: 2015-2016 SLD012 asked 'How much sleep {do you/does SP} usually get at night on weekdays or workdays?' (2 to 14.5 hours); 2017-2018 SLD012 is derived from SLQ300 and SLQ310 (3 to 13.5, with 2 = less than 3 hours and 14 = 14 hours or more), and 2021-2023 SLD012 has the same label. Outcome device: 2015-2018 BP was auscultatory (BPX_I, BPX_J, up to 4 readings); 2021-2023 has only oscillometric BPXO_L (3 readings). Covariate not in 2021-2023: snort or stop breathing (SLQ040). Table 1 oddities: snort or stop breathing '| Yes | 6.72 | 39.08 | 44.22 |  |' and '| No | 41.91 | 8.56 | 5.22 |  |' (the sex columns look swapped against the total), '| Not recorded | 51.37 | 52.36 | 50.56 |  |'; hypertension '| Not recorded | 50.04 | 51.03 | 49.24 |  |'; smoking '| Not recorded | 66.53 | 57.69 | 73.72 |  |'. Software: R and EmpowerStats. The abstract is unstructured; its first estimate is the headline. Figure 1 (flowchart image) was checked against the text and matches; Supplementary Table 1 (DBP) was fetched from the PMC open-access copy to scratchpad/dl/supp/row068/esm1/.",
  "adjudication": null
 },
 {
  "id": "row104",
  "rank": 68,
  "row": 104,
  "doi": "10.1186/s12876-024-03394-6",
  "pmcid": "PMC11378436",
  "title": "Association between neutrophil-to-high-density lipoprotein cholesterol ratio and metabolic dysfunction-associated steatotic liver disease and liver fibrosis in the US population: a nationally representative cross-sectional study using NHANES data from 2017 to 2020",
  "authors": [
   "Lu, Yangni",
   "Xu, Xianli",
   "Wu, Jianlin",
   "Ji, Lei",
   "Huang, Huiya",
   "Chen, Maowei"
  ],
  "year": 2024,
  "journal": "BMC Gastroenterology",
  "table_a": {
   "predictor": "Neutrophil-to-high-density lipoprotein cholesterol ratio",
   "condition": "Metabolic-associated fatty liver conditions",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "We observed a significant positive association between NHR and MASLD (OR = 1.20, 95% CI: 1.09–1.31).",
   "table_location": "Table 2 'Association between NHR and MASLD', column 'Model 3OR (95% CI)', row 'NHR(continuous)'; the same value is Table 4 'Linear effect model' (NAFLD column)",
   "table_quote": "| NHR(continuous) | 1.48(1.38, 1.56) | 1.49 (1.37, 1.62) | 1.20(1.09, 1.31) |",
   "measure": "OR",
   "estimate": 1.2,
   "ci_low": 1.09,
   "ci_high": 1.31,
   "p_value": "not printed",
   "exposure_contrast": "per 1-unit increase in NHR (neutrophils in 10^3 cells/µL divided by HDL-C in mmol/L)",
   "model_label": "Model 3",
   "covariates_in_this_model": [
    "age",
    "race",
    "education",
    "sex",
    "PIR",
    "BMI",
    "diabetes grade",
    "hypertension",
    "smoking",
    "sedentary",
    "total cholesterol",
    "ALT",
    "HbA1c (%)",
    "ALB",
    "SCr",
    "serum uric acid"
   ],
   "n_analytic": 4761,
   "n_quote": "Ultimately, 4,761 individuals were included in the study (Fig. 1).",
   "events": 2123
  },
  "cycles": [
   "2017-2020"
  ],
  "population": {
   "age": "'aged over 20' per Results (exact cutoff unstated; no age criterion is listed among the exclusions)",
   "inclusion": "2017-2020 participants with data to compute NHR, complete elastography (median stiffness and CAP), drinking data without heavy drinking, data on other liver disease causes, and complete covariates",
   "exclusions": [
    "lack of necessary data to calculate NHR (n = 4742)",
    "missing data on median stiffness/CAP or incomplete elastography (n = 2455)",
    "heavy drinking (male average daily drinking > 30 g, female > 20 g) or missing data on drinking (n = 1304)",
    "missing data on other underlying liver disease causes, including autoimmune hepatitis, viral hepatitis infection (HCV-RNA, HCV antibody, or HBsAg positive) and liver cancer (n = 159)",
    "missing data on covariates (n = 2258)"
   ],
   "quote": "This analysis utilized continuous NHANES data from the 2017 to 2020 cycles, involving 15,560 participants. Participants were excluded based on the following criteria: lack of necessary data to calculate NHR (n = 4742), missing data on median stiffness/CAP or incomplete elastography (n = 2455), information on heavy drinking data (defined as male average daily drinking > 30 g, female average daily drinking > 20 g) [24] or missing data on drinking (n = 1304), missing data on other underlying liver disease causes (n = 159), including autoimmune hepatitis, viral hepatitis infection (defined as HCV-RNA, HCV antibody, or HBsAg test positive), liver cancer, and missing data on covariates (n = 2258). Ultimately, 4,761 individuals were included in the study (Fig. 1). ... All participants were aged over 20, with a mean age of 48.22 ± 17.15 years."
  },
  "exposure": {
   "definition": "NHR = neutrophil count (10^3 cells/µL) divided by HDL-C (mmol/L); continuous, per 1 unit.",
   "nhanes_variables": [
    "LBDNENO",
    "LBDHDDSI"
   ],
   "nhanes_files": [
    "CBC",
    "HDL"
   ],
   "transform": "none (continuous, per 1 unit); quartiles in secondary analyses",
   "categories": "Quartiles (Table 2): 'Q1(0.231–2.022)', 'Q2(2.023–2.948)', 'Q3(2.951–4.172)', 'Q4(4.173–16.769)'. The Results text gives different ranges: 'Q1(0.191–2.065), Q2(2.067–2.993), Q3(3.000-4.207) and Q4(4.211–19.028)'. Not used for the headline.",
   "quote": "Neutrophil counts were determined by total blood count using an automated blood analysis system (Coulter DxH 800 analyzer) and displayed as 10 ^ 3 cells / µ l. HDL-C levels were assessed using an automatic device with venous blood samples obtained after 8-hour fasting. The Neutrophil to High-Density Lipoprotein Cholesterol Ratio (NHR) is calculated as the neutrophil count (10^3 cells/µL) divided by the HDL-C level (mmol/L) [28, 29]."
  },
  "outcome": {
   "definition": "MASLD = hepatic steatosis (CAP >= 274 dB/m, grade >= S1), no heavy alcohol consumption (men > 30 g/day, women > 20 g/day), no viral hepatitis, and at least one of: (1) BMI >= 25 kg/m2 or waist >= 94 cm (men) or >= 80 cm (women); (2) fasting glucose >= 100 mg/dL or HbA1c >= 5.7% or type 2 diabetes diagnosis or treatment; (3) BP >= 130/85 mmHg or antihypertensive treatment; (4) triglycerides >= 1.70 mmol/L or lipid-lowering therapy; (5) HDL-C < 1.0 mmol/L (men, or on lipid-lowering therapy) or < 1.3 mmol/L (women).",
   "nhanes_variables": [
    "LUXCAPM",
    "BMXBMI",
    "BMXWAIST",
    "LBXGLU",
    "LBXGH",
    "DIQ010",
    "DIQ050",
    "DIQ070",
    "BPXOSY1",
    "BPXOSY2",
    "BPXOSY3",
    "BPXODI1",
    "BPXODI2",
    "BPXODI3",
    "BPQ040A",
    "BPQ050A",
    "LBDTRSI",
    "BPQ090D",
    "BPQ100D",
    "LBDHDDSI",
    "ALQ121",
    "ALQ130",
    "LBXHCR",
    "LBDHCI",
    "LBDHBG"
   ],
   "nhanes_files": [
    "LUX",
    "BMX",
    "GLU",
    "GHB",
    "DIQ",
    "BPXO",
    "BPQ",
    "TRIGLY",
    "HDL",
    "ALQ",
    "HEPC",
    "HEPBD"
   ],
   "quote": "An influential study has identified that the CAP value (also known as CAP) ≥ 274dB/m, ≥ 290dB/m, and ≥ 302dB/m correspond to hepatic steatosis grades S1, S2, and S3, respectively, with a 90% sensitivity in identifying these conditions [25]. ... The definition of MASLD is based on hepatic steatosis(≥ S1), the absence of heavy alcohol consumption, and no viral hepatitis, along with meeting at least one of the following conditions [24, 27]: (1) Individuals with a BMI ≥ 25 kg/m2 or a waist circumference ≥ 94 cm for men and ≥ 80 cm for women; (2) Fasting glucose levels ≥ 100 mg/dL or hemoglobin A1c levels ≥ 5.7%, or a history of type 2 diabetes diagnosis or current treatment for type 2 diabetes; (3) Blood pressure levels ≥ 130/85 mmHg or current treatment for hypertension; (4) Triglyceride levels ≥ 1.70 mmol/L or individuals on lipid-lowering therapy; (5) Low high-density lipoprotein cholesterol levels, with < 1.0 mmol/L for men or those on lipid-lowering therapy, and < 1.3 mmol/L for women."
  },
  "covariates": [
   {
    "name": "age",
    "coding": "age (years)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "sex",
    "coding": "male/female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "ethnicity",
    "coding": "Mexican American/other Hispanic/non-Hispanic White/non-Hispanic Black/other race",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "education",
    "coding": "below grade 9 / grades 9-11 (including grade 12, no diploma) / high school graduation or GED or equivalent / some college or AA degrees / college degree or above",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "PIR",
    "coding": "Income-to-Poverty Ratio; categories < 1, 1-3.9, and 4 (low, middle, high income)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Whether PIR entered Model 3 as continuous or categories is unstated."
   },
   {
    "name": "smoking status",
    "coding": "never/before/now",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "BMI",
    "coding": "kg/m2; categories < 25, 25-29.9, and 30 kg/m2 (normal, overweight, obese)",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Whether BMI entered Model 3 as continuous or categories is unstated."
   },
   {
    "name": "sedentary behavior",
    "coding": "hour/day (Table 1 shows <6 and ≥ 6)",
    "nhanes_variables": [
     "PAD680"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": true,
    "note": "Whether continuous or dichotomized in Model 3 is unstated."
   },
   {
    "name": "hypertension",
    "coding": "yes/no (definition unstated)",
    "nhanes_variables": [
     "BPXOSY1-3",
     "BPXODI1-3",
     "BPQ020",
     "BPQ040A/BPQ050A (2017-2020)"
    ],
    "nhanes_files": [
     "BPXO",
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "2021-2023 has BPXO_L, BPQ020 and BPQ150."
   },
   {
    "name": "diabetes grade",
    "coding": "Normal/Diabetes/Prediabetes (definition unstated)",
    "nhanes_variables": [
     "DIQ010",
     "DIQ160",
     "LBXGH",
     "LBXGLU"
    ],
    "nhanes_files": [
     "DIQ",
     "GHB",
     "GLU"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "ALT",
    "coding": "IU/L",
    "nhanes_variables": [
     "LBXSATSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "HbA1c",
    "coding": "%",
    "nhanes_variables": [
     "LBXGH"
    ],
    "nhanes_files": [
     "GHB"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "serum albumin (ALB)",
    "coding": "g/dL in Methods; Table 1 reports g/L",
    "nhanes_variables": [
     "LBXSAL",
     "LBDSALSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "serum creatinine (SCr)",
    "coding": "umol/L",
    "nhanes_variables": [
     "LBDSCRSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "total cholesterol",
    "coding": "mmol/L",
    "nhanes_variables": [
     "LBDTCSI (TCHOL) or LBDSCHSI (BIOPRO)"
    ],
    "nhanes_files": [
     "TCHOL",
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Which file is unstated; both exist in 2021-2023."
   },
   {
    "name": "serum uric acid",
    "coding": "µmol/L",
    "nhanes_variables": [
     "LBDSUASI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": ""
   }
  ],
  "design": {
   "weights": "unstated which weight ('Weighting was done following NCHS analytical criteria')",
   "strata_psu": "unstated",
   "quote": "Versions 4.2 and 4.1 of Empowerstats and R were used for all the analyses. Weighting was done following NCHS analytical criteria, and then all data were statistically examined.",
   "software": "EmpowerStats 4.2 and R 4.1",
   "missing_data": "complete case (participants with missing covariates excluded)",
   "quote_missing": "and missing data on covariates (n = 2258)."
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "Utilizing weighted logistics regression analysis, we examined the association between independent and dependent variables. Based on covariate adjustments, three models were developed. Model 1: Unadjusted covariates. Model 2: Adjusted for PIR, age, sex, race, and education. Model 3: Building upon Model 2, further adjusted for uric acid, albumin, creatinine, ALT, hemoglobin A1c, BMI, sedentary behavior, hypertension, diabetes, and cholesterol."
  },
  "unstated": [
   "Which sample weight (e.g. the MEC weight WTMECPRP or a fasting-subsample weight) and whether strata and PSUs were used",
   "The age criterion: Results say all participants were 'aged over 20', but no age exclusion is listed",
   "How average daily alcohol grams were computed (grams per drink; ALQ121 and ALQ130, or dietary recall alcohol)",
   "What 'incomplete elastography' means (exam status, number of valid measures, IQR/median ratio, fasting time)",
   "How missing fasting glucose and triglycerides (measured only in the fasting subsample) were treated in the MASLD cardiometabolic criteria",
   "Which blood pressure readings defined BP >= 130/85 mmHg, and which questionnaire items define current treatment for hypertension, type 2 diabetes diagnosis or treatment, and lipid-lowering therapy",
   "Whether lipid-lowering therapy meets the low-HDL criterion for women as well as men (the sentence is ambiguous)",
   "Whether participants positive for viral hepatitis or with autoimmune hepatitis or liver cancer were excluded from the sample (the flow names only 'missing data on other underlying liver disease causes (n = 159)'), and how autoimmune hepatitis and liver cancer were identified",
   "Definitions of the hypertension and diabetes grade covariates",
   "How each covariate entered Model 3 (continuous or categorical for age, PIR, BMI, sedentary hours) and the reference categories",
   "Which total cholesterol variable (TCHOL file or biochemistry profile)",
   "Whether HDL-C was restricted to fasting participants (the Methods say samples were 'obtained after 8-hour fasting')",
   "Whether smoking is in Model 3 (the Table 2 footnote includes it; the Methods' Model 3 list does not)"
  ],
  "notes": "The exclusion counts (4742 + 2455 + 1304 + 159 + 2258 = 10,918) leave 4,642 of 15,560, not the reported 4,761. Table 1's sex counts add to 4,764 (2267 + 2497). Table 1 prints the Q1 prediabetes count as '1110(8.13%)' and the Q1 normal count as '942((85.01%)' (typos). NHR quartile cutpoints differ between the Results text (Q1 0.191-2.065 ... Q4 4.211-19.028) and Tables 2 and 3 (Q1 0.231-2.022 ... Q4 4.173-16.769). The Results misdescribe the Q4 estimate as 'each unit increase in the NHR ratio was associated with a 3-fold increase'. Table 4 labels the MASLD column 'NAFLD' and its rows 'Model 1' (linear effect) and 'Model 2' (piecewise); its linear-effect OR equals the Table 2 Model 3 OR. Model 3's covariate list differs between the Methods (no smoking) and the Table 2 footnote (with smoking); the footnote list is recorded. 15,560 is the size of the 2017-March 2020 pre-pandemic sample, so the P files and their weights (e.g. WTMECPRP) apply (extractor's inference). The abstract's next estimate (OR = 1.01; 95% CI: 0.94-1.09) is for liver fibrosis, a different outcome. NHANES variable names are the extractor's mapping; the paper names none (BPQ040A/BPQ050A and BPQ090D/BPQ100D are the 2017-2020 medication items; 2021-2023 has BPQ150 and BPQ101D). Software: EmpowerStats 4.2 and R 4.1. Europe PMC lists no erratum (only a preprint link).",
  "adjudication": null
 },
 {
  "id": "row218",
  "rank": 69,
  "row": 218,
  "doi": "10.1038/s41598-024-51216-2",
  "pmcid": "PMC10767076",
  "title": "Association between weight-adjusted-waist index and urge urinary incontinence: a cross-sectional study from NHANES 2013 to 2018",
  "authors": [
   "Sun, Haohao",
   "Huang, Jingxi",
   "Tang, Hao",
   "Wei, Bingbing"
  ],
  "year": 2024,
  "journal": "Scientific Reports",
  "table_a": {
   "predictor": "Weight-adjusted-waist index",
   "condition": "Urinary incontinence",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "Overall UUI was more prevalent with elevated WWI (OR 1.20, 95% CI 1.13–12.8, P < 0.0001), which similar results were observed in weekly (OR 1.32, 95% CI 1.18–1.48, P < 0.0001) and daily (OR 1.27, 95% CI 1.06–1.53, P = 0.0091) UUI.",
   "table_location": "Table 2 'Association between weight-adjusted-waist index and UUI', block 'Overall UUI (ORa, 95% CIb, P)', row 'Model 3e' (superscript e marks the footnote), column 'WWI continuous'; the same value is Table 3 row 'Weight-adjusted-waist index' and Table 4 'Fitting by standard linear model' (WWI)",
   "table_quote": "| Model 3e | Ref | 1.08 (0.93, 1.25)0.3242 | 1.23 (1.06, 1.43)0.0069 | 1.47 (1.26, 1.72)< 0.0001 | < 0.0001 | 1.20 (1.13, 1.28)< 0.0001 |",
   "measure": "OR",
   "estimate": 1.2,
   "ci_low": 1.13,
   "ci_high": 1.28,
   "p_value": "< 0.0001",
   "exposure_contrast": "per 1-unit increase in WWI (cm/sqrt(kg))",
   "model_label": "Model 3",
   "covariates_in_this_model": [
    "gender",
    "age",
    "race",
    "education level",
    "poverty income ratio",
    "health insurance coverage",
    "alcohol use",
    "diabetes",
    "smoke"
   ],
   "n_analytic": 14118,
   "n_quote": "our final analysis comprised 14,118 eligible participants (as shown in Fig. 1).",
   "events": 3338
  },
  "cycles": [
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=18 as stated (the urge-leakage question is asked only at ages 20+, so the analytic sample is in effect 20+)",
   "inclusion": "NHANES 2013-2018 participants with complete UUI and WWI data",
   "exclusions": [
    "under 18 years old (n = 11,439)",
    "missing data related to pregnancy (n = 190)",
    "missing UUI (n = 3040)",
    "missing WWI (n = 613)"
   ],
   "quote": "Our analysis consisted of participants who had complete data on both UUI and WWI. Initially, 29,400 individuals were included in the study, and after excluding individuals under 18 years old (n = 11,439) and those with missing data related to pregnancy (n = 190), UUI (n = 3040), and WWI (n = 613), our final analysis comprised 14,118 eligible participants (as shown in Fig. 1)."
  },
  "exposure": {
   "definition": "Weight-adjusted-waist index = waist circumference (cm) divided by the square root of weight (kg); continuous, per 1 unit. The Methods text inverts this ('the square root of their waist circumference (cm) and dividing it by their weight (kg)'), but the reported WWI range (8.46 to 14.79) and the Table 1 means fit WC/sqrt(weight).",
   "nhanes_variables": [
    "BMXWAIST",
    "BMXWT"
   ],
   "nhanes_files": [
    "BMX"
   ],
   "transform": "none (continuous, per 1 unit); quartiles in a sensitivity analysis",
   "categories": "Quartiles (Results, Baseline characteristics): 'WWI quartiles were defined as follows: Q1 (8.46–10.54), Q2 (10.54–11.12), Q3 (11.12–11.70), and Q4 (11.70–14.79).' Not used for the headline.",
   "quote": "WWI is a method of measuring obesity that utilizes waist circumference and weight. Higher WWI scores indicate a higher level of obesity. Weight and WC measurements were taken in a mobile examination center (MEC) by certified health technicians. The WWI of every participant was determined by taking the square root of their waist circumference (cm) and dividing it by their weight (kg). Participants were grouped for subsequent analyses based on their WWI quartiles, and WWI was also considered a continuous variable."
  },
  "outcome": {
   "definition": "Overall UUI = 'yes' to the past-12-month question on leaking urine with an urge or pressure to urinate before reaching the toilet. Weekly and daily UUI (secondary outcomes) use the follow-up frequency question.",
   "nhanes_variables": [
    "KIQ044",
    "KIQ450 (frequency; weekly and daily UUI only)"
   ],
   "nhanes_files": [
    "KIQ_U"
   ],
   "quote": "UUI was determined by the response to the question, “During the past 12 months, have you leaked or lost control of even a small amount of urine with an urge or pressure to urinate and you could not get to the toilet fast enough?” UUI severity was characterized by the response to the question, “How frequently does this occur?” At least weekly UUI and at least daily UUI were characterized as variables separate from any overall UUI."
  },
  "covariates": [
   {
    "name": "gender",
    "coding": "male/female (Table 3: 'Female (vs. male)')",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "age",
    "coding": "continuous, per year (Table 3: 'Age(year)')",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "race",
    "coding": "Mexican American (reference), Other Hispanic, Non-Hispanic White, Non-Hispanic Black, Other races",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "education level",
    "coding": "less than high school (reference), high school or GED, above high school",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "poverty income ratio",
    "coding": "continuous (Table 3: 'Poverty income ratio')",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "health insurance coverage",
    "coding": "yes (reference)/no: 'Are you covered by health insurance or some other kind of health care plan?'",
    "nhanes_variables": [
     "HIQ011"
    ],
    "nhanes_files": [
     "HIQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "alcohol use",
    "coding": "yes (reference)/no: having consumed a minimum of 12 standard alcoholic drinks within a year",
    "nhanes_variables": [
     "ALQ101 (2013-2016)"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": false,
    "note": "The '12 drinks in a year' item (ALQ101) is not in ALQ_L (it was already dropped in 2017-2018, so how the paper coded 2017-2018 is unstated); ALQ111 (ever had a drink) and ALQ121 (past-12-month frequency) are the nearest items."
   },
   {
    "name": "diabetes",
    "coding": "yes (reference)/no (definition unstated)",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "smoke",
    "coding": "yes (reference)/no: yes if a positive response to 'Have you smoked at least 100 cigarettes in your entire life?' or 'Do you now smoke cigarettes?'",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": ""
   }
  ],
  "design": {
   "weights": "NHANES sampling weights; which weight (e.g. the MEC exam weight) and how it was combined across the three cycles is unstated",
   "strata_psu": "unstated (the design is described as accounting for 'multistage cluster surveys')",
   "quote": "The guidelines provided by the Centers for Disease Control and Prevention (CDC) were used for all statistical analyses in this study and incorporated NHANES sampling weights to address the complexities of multistage cluster surveys. ... Both R (version 4.2.0) and EmpowerStats (version 4.1) were used for all statistical analyses and statistical significance with a two-sided P value < 0.05 was defined.",
   "software": "R (version 4.2.0) and EmpowerStats (version 4.1)",
   "missing_data": "imputation (mode for categorical and median for continuous variables)",
   "quote_missing": "In order to handle missing values, the mode and median of the cases that already had missing data for categorical variables and continuous variables, respectively, were imputed."
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "Weighted multivariate logistic regression and generalized additive models were used to investigate the connection between WWI and UUI and its nonlinearity. ... To investigate the relationship between WWI and UUI, three distinct models were analyzed using weighted multivariable regression. Model 1 did not involve any modifications for covariates, while Model 2 incorporated adjustments for age, gender, and race. Model 3 was further adjusted for age, gender, race, education level, poverty-income ratio, health insurance coverage, alcohol use, diabetes, and smoking."
  },
  "unstated": [
   "Which weight (the urology items were a MEC self-interview, so presumably the MEC exam weight) and how it was combined across the three cycles",
   "Whether strata and PSUs were used for variance estimation",
   "What 'missing data related to pregnancy (n = 190)' means: excluding pregnant participants or those with missing pregnancy status",
   "How 'Don't know' and 'Refused' answers to the urge-leakage question were handled",
   "Diabetes definition (self-report only, or also glucose, HbA1c or medication)",
   "How 'at least 12 alcoholic drinks within a year' was determined for 2017-2018, where that question was not asked",
   "How the five NHANES education levels were collapsed into three",
   "Which covariates had values imputed, and how many",
   "The age criterion in force (stated >= 18, but the urge-leakage question targets ages 20+)"
  ],
  "notes": "The text file has no abstract; it was read from the Europe PMC XML already downloaded (scratchpad/dl/fulltext/row218_PMC10767076.xml) and matches the Scopus abstract in Table A. The abstract prints the CI as '1.13–12.8', a typo for 1.28: Tables 2, 3 and 4 and the Results text give 1.13 to 1.28, so ci_high is taken from the table. The Methods misstate the WWI formula as sqrt(WC)/weight; the WWI values (8.46 to 14.79) and the ratio of Table 1 means (Q1: 86.44 cm / sqrt(74.94 kg) = 9.99; Q4: 116.26 / sqrt(92.59) = 12.08) fit WC/sqrt(weight). The starting 29,400 equals the full 2013-2014 (10,175), 2015-2016 (9,971) and 2017-2018 (9,254) samples (extractor's inference from NHANES file sizes); the exclusion counts add up (29,400 - 11,439 - 190 - 3040 - 613 = 14,118). Table 1's title says 'tertiles' but it shows quartiles. The Results say '838 (5.94%) reported at least weekly SUI' (presumably UUI). The abstract says P > 0.05 for all interactions while the subgroup text says 'all P values for interaction < 0.05'. The Results text assigns 25%, 10% and 39% lower odds to Non-Hispanic White, Other Hispanic and Other races, which does not match Table 3 (Other Hispanic 0.75, Non-Hispanic White 0.90, Other races 0.61). BMI and WC are listed as covariates but are not in Model 3. Table 3 gives every Model 3 coefficient (e.g. Female vs male 2.45 (2.21, 2.71); age per year 1.04 (1.04, 1.04); Diabetes no vs yes 0.70 (0.62, 0.79)), useful for checking a re-run. The 2021-2023 urology documentation says the adult questions did not change from 2017-March 2020 apart from removed items (KIQ026, KIQ029, KIQ430 (printed 'KIQ0430'), KIQ450, KIQ046, KIQ470, KIQ050), so KIQ044 is unchanged. NHANES variable names are the extractor's mapping; the paper names none. Software: R 4.2.0 and EmpowerStats 4.1. Europe PMC lists no erratum.",
  "adjudication": null
 },
 {
  "id": "row306",
  "rank": 70,
  "row": 306,
  "doi": "10.3390/nu15051177",
  "pmcid": "PMC10004774",
  "title": "Association between Systemic Immunity-Inflammation Index and Hyperlipidemia: A Population-Based Study from the NHANES (2015–2020)",
  "authors": [
   "Mahemuti, Nayili",
   "Jing, Xiyue",
   "Zhang, Naijian",
   "Liu, Chuanlang",
   "Li, Changping",
   "Cui, Zhuang",
   "Liu, Yuanyuan",
   "Chen, Jiageng"
  ],
  "year": 2023,
  "journal": "Nutrients",
  "table_a": {
   "predictor": "Systemic immune-inflammation index",
   "condition": "Hyperlipidemia",
   "population": "US population"
  },
  "headline": {
   "abstract_quote": "A substantial positive correlation between SII and hyperlipidemia was found [1.03 (1.01, 1.05)] in a multivariate linear regression analysis.",
   "table_location": "Table 3 ('The association between SII and hyperlipidemia.'), column 'PartiallyAdjusted Model(Model 2)', sub-header 'OR (95% CI) p-Value', row 'SII/100'. The abstract gives only this one estimate; it matches Model 2 uniquely (Model 1 is 1.04 (1.02, 1.06), Model 3 is 1.02 (1.00, 1.04)).",
   "table_quote": "| SII/100 | 1.04 (1.02, 1.06) *** | 1.03 (1.01, 1.05) * | 1.02 (1.00, 1.04) |",
   "measure": "OR",
   "estimate": 1.03,
   "ci_low": 1.01,
   "ci_high": 1.05,
   "p_value": "* (Table 3 footnote: '* p < 0.05')",
   "exposure_contrast": "per 100-unit increase in SII (SII/100), where SII = platelet count x neutrophil count / lymphocyte count with counts in 10^3 cells/uL",
   "model_label": "Model 2 (PartiallyAdjusted Model)",
   "covariates_in_this_model": [
    "age",
    "sex",
    "race"
   ],
   "n_analytic": 6117,
   "n_quote": "[Abstract] A total of 6117 US adults were included in our study. [2.1. Study Population] The research included a total of 6117 individuals. [comment: Table 3 does not print the number of participants in each model.]",
   "events": 4265
  },
  "cycles": [
   "2015-2016",
   "2017-March 2020 (pre-pandemic)"
  ],
  "population": {
   "age": ">=20",
   "inclusion": "NHANES 2015-2020 participants with complete SII and hyperlipidemia data, aged 20 years or older",
   "exclusions": [
    "missing SII data: 5264 (25,531 to 20,267 in Fig. 1)",
    "missing hyperlipidemia data: 12,969 (20,267 to 7,298 in Fig. 1)",
    "younger than 20 years: 1181 (7,298 to 6,117 in Fig. 1)"
   ],
   "quote": "[2.1. Study Population] In the investigation, we removed from the 25,531 eligible people 5264 participants with missing SII data, 12,969 participants with missing hyperlipidemia data, and 1181 participants younger than 20 years of age. The research included a total of 6117 individuals. Figure 1 depicts the sample selection. [Fig. 1 (image), box text: Participant with SII data (N=20267); Participants with hyperlipidemia data (N=7298); Final participants (N=6117)]"
  },
  "exposure": {
   "definition": "Systemic immune-inflammation index = platelet count x neutrophil count / lymphocyte count, counts from the CBC (Coulter DxH 800) in 10^3 cells/uL (printed as '103 cells/mL', superscript lost). The headline model enters SII/100, so the OR is per 100 SII units.",
   "nhanes_variables": [
    "LBXPLTSI",
    "LBDNENO",
    "LBDLYMNO"
   ],
   "nhanes_files": [
    "CBC (CBC_I, P_CBC; CBC_L in 2021-2023)"
   ],
   "transform": "per 100 units (SII/100); quartiles only in the sensitivity analysis; raw SII in the threshold (two-segment) model",
   "categories": "[comment: quartiles Q1 to Q4 (Q1 reference) in the sensitivity analysis only; cutpoints not reported.] [Table 2, header row] |  | N = 1529 | N = 1529 | N = 1529 | N = 1530 |  |",
   "quote": "[Abstract] SII was computed by dividing the platelet count × the neutrophil count/the lymphocyte count. [2.3. SII and Covariates] Using automated hematology analysis equipment (a CoulterDxH 800 analyzer), the lymphocyte, neutrophil, and platelet counts were measured and reported as 103 cells/mL. The SII level was determined by multiplying the platelet count by the neutrophil count/lymphocyte count [7,22,23]. [3.2. Association between SII and Hyperlipidemia] Because the effect value is not apparent, SII/100 is used to amplify the effect value by 100 times."
  },
  "outcome": {
   "definition": "Hyperlipidemia (NCEP ATP III): total cholesterol 200 mg/dL, triglycerides 150 mg/dL, HDL 40 mg/dL in males and 50 mg/dL in females, or LDL 130 mg/dL, or self-reported use of cholesterol-lowering drugs. The published text (Europe PMC XML and the publisher PDF) prints no inequality signs; the cited ATP III convention is TC >= 200, TG >= 150, HDL < 40 (men) or < 50 (women), LDL >= 130 mg/dL.",
   "nhanes_variables": [
    "LBXTC",
    "LBXTR (LBXTLG in 2021-2023)",
    "LBDHDD",
    "LBDLDL",
    "BPQ100D (BPQ101D in 2021-2023)",
    "RIAGENDR"
   ],
   "nhanes_files": [
    "TCHOL",
    "TRIGLY",
    "HDL",
    "BPQ",
    "DEMO"
   ],
   "quote": "[2.2. Assessment of Hyperlipidemia] Adult Treatment Panel III (ATP 3) of the National Cholesterol Education Program (NCEP) classified hyperlipidemia as total cholesterol 200 mg/dL, triglycerides 150 mg/dL, HDL 40 mg/dL in males and 50 mg/dL in females, or low-density lipoprotein 130 mg/dL [21]. Alternately, persons who reported using cholesterol-lowering drugs were also classified as having hyperlipidemia."
  },
  "covariates": [
   {
    "name": "age",
    "coding": "unstated (Table 1 reports mean ± SD)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In the headline Model 2."
   },
   {
    "name": "sex",
    "coding": "men, women",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In the headline Model 2."
   },
   {
    "name": "race",
    "coding": "Mexican American, Non-Hispanic White, Non-Hispanic Black, other Hispanic, other race",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In the headline Model 2. Variable not named; the five groups match RIDRETH1."
   },
   {
    "name": "education level",
    "coding": "less than high school, high school, more than high school",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only. Mapping of DMDEDUC2 codes to the three groups unstated."
   },
   {
    "name": "income-to-poverty ratio",
    "coding": "three categories printed as '1.5, 1.5–3.5, and >3.5' (Table 1 rows: 0–1.5, 1.5–3.5, >3.5)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only."
   },
   {
    "name": "marital status",
    "coding": "married/living with a partner, widowed/divorced/separated, never married",
    "nhanes_variables": [
     "DMDMARTL (2015-2016)",
     "DMDMARTZ (2017-2020 and 2021-2023)"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only. DMDMARTZ in 2021-2023 has exactly these three groups."
   },
   {
    "name": "drinking status",
    "coding": "excessive (three drinks per day for women, four for men), moderate (two drinks per day for women, three for men), light (all other); inequality directions and the coding of non-drinkers unstated",
    "nhanes_variables": [
     "ALQ130"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only. Variables not named; ALQ130 (average drinks per drinking day) exists in all three periods."
   },
   {
    "name": "smoking status",
    "coding": "never (no more than 100 cigarettes in life), former (more than 100, not smoking now), current (more than 100, smoking sometimes or consistently)",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only."
   },
   {
    "name": "BMI",
    "coding": "<25, 25 to 30, >30 kg/m2",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only."
   },
   {
    "name": "hypertension",
    "coding": "average systolic >140 mmHg and/or diastolic 90 mmHg (direction for diastolic not printed), physician-diagnosed hypertension, or antihypertensive medication",
    "nhanes_variables": [
     "BPXSY1-BPXSY4 and BPXDI1-BPXDI4 (2015-2016, auscultatory)",
     "BPXOSY1-BPXOSY3 and BPXODI1-BPXODI3 (2017-2020 and 2021-2023, oscillometric)",
     "BPQ020",
     "BPQ050A (BPQ150 in 2021-2023)"
    ],
    "nhanes_files": [
     "BPX",
     "BPXO",
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only. 2021-2023 has only oscillometric readings (BPXO_L)."
   },
   {
    "name": "diabetes",
    "coding": "reported diabetes diagnosis and ('and' as printed) use of diabetes medicine or insulin",
    "nhanes_variables": [
     "DIQ010",
     "DIQ050",
     "DIQ070"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only."
   }
  ],
  "design": {
   "weights": "unstated: the paper says only 'We used a weighting strategy'; which weight (MEC, fasting subsample, or the 2017-March 2020 pre-pandemic weights) and how the 2015-2016 and 2017-2020 files were combined are not given",
   "strata_psu": "unstated",
   "quote": "[2.4. Statistical Analysis] the differences between participants grouped by SII quartiles and the differences between participants with or without hyperlipidemia were assessed using a weighted t-test (continuous variables) or a weighted chi-square test (categorical variables). ... We used a weighting strategy to lessen the substantial volatility of our dataset.",
   "software": "R studio (Version 4.2.2) and EmpowerStats (version 2.0)",
   "missing_data": "complete case for SII and hyperlipidemia (participants missing either were excluded); handling of missing covariates unstated (it matters only for Model 3: the headline Model 2 uses age, sex, and race, which have no missing values in NHANES demographics)",
   "quote_missing": "[2.1. Study Population] In the investigation, we removed from the 25,531 eligible people 5264 participants with missing SII data, 12,969 participants with missing hyperlipidemia data, and 1181 participants younger than 20 years of age."
  },
  "model": {
   "family": "logistic",
   "weighted": null,
   "quote": "[2.4. Statistical Analysis] To examine the association between SII and hyperlipidemia, multivariate logistic regression analysis between SII and hyperlipidemia was used to construct multivariate tests, using three models with no covariates in model 1; model 2 was adjusted for age, sex, and race; model 3 was adjusted for age, sex, race, marital status, income to poverty ratio, education level, drinking status, smoking status, BMI, hypertension, and diabetes; and SII and hyperlipidemia were evaluated using odds ratios (OR) and 95% confidence interval (CI) in the models. [comment: whether the regressions themselves were survey-weighted is not stated; the only statement is the next quote.] We used a weighting strategy to lessen the substantial volatility of our dataset."
  },
  "unstated": [
   "Which NHANES files: the paper says only '2015–2020' (25,531 matches 2015-2016 plus 2017-March 2020 pre-pandemic)",
   "Which survey weight (MEC, fasting subsample, or the pre-pandemic WTMECPRP/WTSAFPRP-type weights), how the two periods' weights were combined, and whether strata (SDMVSTRA) and PSU (SDMVPSU) were used",
   "Whether the Table 3 logistic regressions were survey-weighted at all",
   "Inequality directions of the lipid thresholds (no signs in the published text) and of the diastolic hypertension threshold",
   "Whether hyperlipidemia required all four lipid measures (which would restrict to the fasting subsample; the 12,969 exclusions for missing hyperlipidemia data suggest so) or was assigned when any one criterion was met",
   "Which LDL-C variable (2015-2016 has only Friedewald LBDLDL; 2017-2020 also has Martin-Hopkins LBDLDLM and NIH equation 2 LBDLDLN)",
   "Which questionnaire item defined cholesterol-lowering drug use (presumably BPQ100D) and how participants skipped past it were coded",
   "Whether age entered the models as continuous or in categories",
   "Reference categories for sex and race",
   "Whether pregnant women, participants on medications affecting blood counts, or outlying SII values were excluded",
   "Handling of missing covariates in Model 3 (no covariate-based exclusions are listed)",
   "Number of participants in each Table 3 model"
  ],
  "notes": "Headline is the abstract's only estimate, 1.03 (1.01, 1.05), which is Model 2 (age, sex, race) in Table 3; the fully adjusted Model 3 estimate is 1.02 (1.00, 1.04) and the text calls it insignificant. The abstract calls the model 'multivariate linear regression' and Table 4's title says 'linear regression model', but both report ORs from logistic models (EmpowerStats wording). Section 2.3 contradicts itself ('The systemic immunity-inflammation index is the dependent variable in this investigation. SII was intended as an exposure variable in our research.'); the tables treat SII as the exposure. Units printed as '103 cells/mL' are 10^3 cells/uL (superscript lost; mean SII 459.54 ± 317.28 fits that unit). The flowchart (Fig. 1, checked as an image from PMC) matches the text: 25,531 to 20,267 to 7,298 to 6,117. The large loss for missing hyperlipidemia data (12,969) suggests the fasting-subsample triglyceride and LDL values were required, which would make the fasting-subsample weight the appropriate one, but the paper does not say. Odd values elsewhere: Table 4 women 'SII < 958.14 | 1.0006 (1.0002, 1.1010)' has an implausible upper bound (likely 1.0010); Table 2's 'Light alcohol consumption' row repeats the values of the smoking 'Never' row. Software: R 4.2.2 and EmpowerStats 2.0. In 2021-2023 the triglyceride variable is LBXTLG (not LBXTR) and the cholesterol-medication item is BPQ101D (asked with a different skip pattern than BPQ100D).",
  "adjudication": null
 },
 {
  "id": "row221",
  "rank": 72,
  "row": 221,
  "doi": "10.1016/j.heliyon.2024.e27764",
  "pmcid": "PMC10950664",
  "title": "Association between Sitting Time and Urinary Incontinence in the US population: Data from the National Health and Nutrition Examination Survey (NHANES) 2007 to 2018",
  "authors": [
   "Di, Xingpeng",
   "Yuan, Chi",
   "Xiang, Liyuan",
   "Wang, Guanbo",
   "Liao, Banghua"
  ],
  "year": 2024,
  "journal": "Heliyon",
  "table_a": {
   "predictor": "Sedentary behavior",
   "condition": "Urinary incontinence",
   "population": "US population"
  },
  "headline": {
   "abstract_quote": "Prolonged sitting time was associated with urgency UI (UUI, odds ratio [OR] = 1.2, 95% confidence interval [CI] = 1.1 to 1.3, p = 0.001).",
   "table_location": "Table 2 ('Logistic regression analyses of the association between UI and sitting time by sex-stratified linear regression model, weighted.'), outcome block 'UUI', sitting time '≥ 7' (reference '< 7'), row 'Model 1', column 'All (OR, 95% CI), p'. It is the only cell with 1.2 (1.1,1.3), 0.001; the male Model 1 cell is 1.2 (0.99,1.4), 0.1.",
   "table_quote": "| Model 1 <colspan=2> | 1.2 (1.1,1.3), 0.001** | 1.2 (0.99,1.4), 0.1 | 1.2 (1.1,1.4), 0.003** |",
   "measure": "OR",
   "estimate": 1.2,
   "ci_low": 1.1,
   "ci_high": 1.3,
   "p_value": "p = 0.001 (Abstract); 0.001** (Table 2)",
   "exposure_contrast": "sitting time >= 7 h per day vs < 7 h per day (reference)",
   "model_label": "Model 1",
   "covariates_in_this_model": [
    "age",
    "race",
    "education level",
    "family income-to-poverty ratio",
    "marital state"
   ],
   "n_analytic": 22916,
   "n_quote": "[Abstract] A total of 22,916 participants were enrolled. [Results] A total of 22,916 adult participants (10,768 females and 12,148 males) were enrolled for further analysis (Fig. 1). [comment: the number of participants in the UUI regression is not printed; if UUI was compared only with participants without any UI, it would be smaller than 22,916.]",
   "events": 2951
  },
  "cycles": [
   "2007-2008",
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=20",
   "inclusion": "NHANES 2007-2018 participants aged 20 and above with complete data on UI, sitting time, and covariates",
   "exclusions": [
    "younger than 20 years: n = 25,372 (Fig. 1 then shows 34,770 adults, although 59,842 - 25,372 = 34,470)",
    "did not respond to the survey on UI symptoms and sitting time: n = 4863 (Fig. 1: 29,907 remain)",
    "unknown covariates: n = 6991 (Fig. 1: 22,916 remain)"
   ],
   "quote": "[Study population] The exclusion criteria were as follows: (1) individuals younger than 20 years of age (n = 25,372); (2) individuals who did not respond to the survey on UI symptoms and sitting time (n = 4863); (3) individuals with unknown covariates (n = 6991). [Fig. 1 (image), box text: Overall participants in NHANES 2007-2018 (n = 59842); Age < 20 years old (n = 25372); Complete information for adults (n = 34770); Missing sitting time data Missing UI data (n = 4863); Complete information for UI and sitting time (n = 29907); Missing covariates (n = 6991); Included for analysis (n = 22916); Male participants (n =12148); Female participants (n =10768)]"
  },
  "exposure": {
   "definition": "Self-reported time spent sitting on a typical day (NHANES sedentary-activity item PAD680, recorded in minutes), dichotomized as < 7 h vs >= 7 h per day.",
   "nhanes_variables": [
    "PAD680"
   ],
   "nhanes_files": [
    "PAQ (PAQ_E to PAQ_J; PAQ_L in 2021-2023)"
   ],
   "transform": "dichotomized at 7 h per day (420 minutes)",
   "categories": "[Definition of sitting time] The sitting time was divided into “< 7 hours (h)” and “≥ 7 h” according to previous studies [9]. [comment: < 7 h is the reference in Table 2.]",
   "quote": "[Definition of sitting time] Sitting time is an important component of sedentary activities and was collected by a self-report questionnaire. The NHANES defined sitting time by the question “How much time spent sitting or reclining on a typical day”. The sitting time was divided into “< 7 hours (h)” and “≥ 7 h” according to previous studies [9]."
  },
  "outcome": {
   "definition": "Urgency urinary incontinence (UUI): 'yes' to the past-12-month urge leakage question (KIQ044). The paper defines SUI by the stress leakage question (KIQ042) and MUI as 'yes' to both. Table 1 and the Results counts treat no UI, SUI, UUI and MUI as mutually exclusive groups, so 'UUI' there means urge leakage without stress leakage; the regression's outcome coding and comparison group are not stated.",
   "nhanes_variables": [
    "KIQ044",
    "KIQ042"
   ],
   "nhanes_files": [
    "KIQ_U (KIQ_U_E to KIQ_U_J; KIQ_U_L in 2021-2023)"
   ],
   "quote": "[Definition of urinary incontinence symptoms] The “Kidney Condition-Urology” questionnaire defined SUI by the question “During the past 12 months, have you leaked or lost control of even a small amount of urine with an activity like coughing, lifting or exercise?“. UUI was determined by the question “During the past 12 months, have you leaked or lost control of even a small amount of urine with an urge or pressure to urinate and you couldn't get to the toilet fast enough like coughing, lifting, or exercise?“. MUI was determined based on “yes” answers to both the SUI and UUI questions."
  },
  "covariates": [
   {
    "name": "age",
    "coding": "unstated (Table 1 reports mean ± SD)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In the headline Model 1."
   },
   {
    "name": "race",
    "coding": "Mexican American, non-Hispanic Black, non-Hispanic White, other Hispanic, other races",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In the headline Model 1. Variable not named; the five groups match RIDRETH1."
   },
   {
    "name": "education level",
    "coding": "less than 12th grade, high school graduate, college graduate (Table 1: Lower than 12th grade, High school grade, College grade)",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In the headline Model 1. Mapping of DMDEDUC2 codes unstated (whether 'college graduate' means some college or above)."
   },
   {
    "name": "family income-to-poverty ratio",
    "coding": "<1.3, 1.3–3.5, >3.5 (Table 1: <1.3, ≥1.3 <3.5, ≥ 3.5)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In the headline Model 1."
   },
   {
    "name": "marital status",
    "coding": "Table 1 shows six groups: Married, Divorced, Widowed, Separated, Living with partner, Never married; coding in the models unstated",
    "nhanes_variables": [
     "DMDMARTL"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": false,
    "note": "In the headline Model 1. 2021-2023 has only DMDMARTZ with three groups (married/living with partner, widowed/divorced/separated, never married), so the six-group coding cannot be rebuilt; a collapsed three-group version can."
   },
   {
    "name": "BMI",
    "coding": "weight/height2, kg/m2 (Table 1 groups: ≤ 20, >20 ≤25, >25 ≤30, >30); model coding unstated",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Model 2 only."
   },
   {
    "name": "smoking history",
    "coding": "Table 1: Non-smoker, Smoker; definition unstated",
    "nhanes_variables": [
     "SMQ020"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "Model 2 only. Variable not named; SMQ020 (100 cigarettes in life) is the usual source."
   },
   {
    "name": "alcohol consumption history",
    "coding": "<1 drink per week, 1–3 drinks per week, >4 or more drinks per week (Table 1: < 1, 1–3, ≥ 4 drinks/week)",
    "nhanes_variables": [
     "ALQ120Q",
     "ALQ120U",
     "ALQ130",
     "ALQ121"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "Model 2 only. Derivation unstated; 2007-2016 use ALQ120Q/ALQ120U with ALQ130, 2017-2018 and 2021-2023 use ALQ121 with ALQ130."
   },
   {
    "name": "moderate recreational activity",
    "coding": "yes/no to: In a typical week do you do any moderate-intensity sports, fitness, or recreational activities ... for at least 10 min continuously?",
    "nhanes_variables": [
     "PAQ665 (2007-2018)",
     "PAD790Q/PAD790U (2021-2023)"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": true,
    "note": "Model 2 only. 2021-2023 asks how often moderate leisure-time activity is done (PAD790Q/U) instead of the yes/no item and has no 10-minute minimum; 'any' can be derived as frequency > 0."
   },
   {
    "name": "vigorous recreational activity",
    "coding": "yes/no to: In a typical week do you do any vigorous-intensity sports, fitness, or recreational activities ... for at least 10 min continuously?",
    "nhanes_variables": [
     "PAQ650 (2007-2018)",
     "PAD810Q/PAD810U (2021-2023)"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": true,
    "note": "Model 2 only. Same wording change as for moderate activity."
   },
   {
    "name": "DM",
    "coding": "yes/no; definition unstated",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "Model 2 only."
   },
   {
    "name": "hypertension",
    "coding": "yes/no; definition unstated",
    "nhanes_variables": [
     "BPQ020"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Model 2 only."
   },
   {
    "name": "coronary heart disease",
    "coding": "yes/no; definition unstated",
    "nhanes_variables": [
     "MCQ160C"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "Model 2 only. MCQ160c in MCQ_L."
   }
  ],
  "design": {
   "weights": "unstated which weight (the UI items are MEC self-interview items, sitting time is from the household interview, BMI from the exam) or how six cycles were combined; the paper says only that CDC-recommended sampling weights were used",
   "strata_psu": "stated as used (CDC-recommended strata and primary sample units); variables not named",
   "quote": "[Statistical analysis] The sampling weights, strata, and primary sample units were recommended by the Centers for Disease Control and Prevention (CDC) to better represent the US population.",
   "software": "R software version 4.1 and EmpowerStats (X&Y Solutions)",
   "missing_data": "complete case",
   "quote_missing": "[Statistical analysis] Missing data were excluded for complete case analyses."
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "[Statistical analysis] Furthermore, weighted multivariate logistic regression analyses were adopted to assess the association between sitting time and UI in models 1 and 2. The crude model was not adjusted. Model 1 was adjusted for baseline demographic information, including age, race, education level, family income-to-poverty ratio, and marital status. [Results] Weighted logistic regression analyses were performed to identify the relationship between sitting time and UI symptoms (Table 2)."
  },
  "unstated": [
   "Which survey weight (MEC exam weight WTMEC2YR fits MEC self-interview UI items; interview weight WTINT2YR otherwise) and how weights were combined across six cycles (e.g. divided by 6)",
   "Coding of the UUI outcome in the regression: urge leakage without stress leakage (the paper's mutually exclusive UUI group) or any urge leakage (KIQ044 = yes, including mixed)",
   "Comparison group for the UUI regression: participants with no UI only, or all participants without the UUI outcome",
   "Handling of 'refused'/'don't know' answers to KIQ042/KIQ044 and of PAD680 codes 7777/9999",
   "Whether age entered the models as continuous or in categories",
   "Mapping of DMDEDUC2 into the three education groups",
   "Coding of marital status in the models (six groups in Table 1)",
   "How BMI entered the models (continuous or the Table 1 groups)",
   "Definitions of smoking history, DM, hypertension, and coronary heart disease",
   "How drinks per week were derived and where the boundary between 1–3 and '>4 or more' falls (3 to 4 drinks)",
   "Reference categories for all categorical covariates",
   "Whether pregnant women were excluded",
   "Sex is not listed among the covariates for the 'All' column in any model; whether that is accurate is not stated",
   "Number of participants in each regression"
  ],
  "notes": "Headline choice: Suchak et al.'s condition is 'Urinary incontinence'; the abstract's first estimate with a CI is for urgency UI, a UI subtype, in the whole population, so it is the headline under the brief's rule. The abstract gives no total-UI estimate; in Table 2 total UI is null in every model for 'All' (Model 2: 1.0 (0.9,1.1), 0.8). The Results text attributes the headline numbers to males ('among males'), but Table 2 places 1.2 (1.1,1.3), 0.001** in the 'All' column (the male Model 1 cell is 1.2 (0.99,1.4), 0.1) and the Conclusions say 'in all populations'; the 'All' reading is used. The headline is Model 1 (demographics only) because it is the estimate the abstract reports; the fully adjusted Model 2 estimate for UUI in 'All' is 1.1 (1.0,1.2), 0.03*. Estimates are printed to one decimal place. Neither model lists sex as a covariate. The UUI question as quoted in Methods ends with 'like coughing, lifting, or exercise', copied from the stress item; the NHANES KIQ044 text has no such phrase. Flow inconsistency: 59,842 - 25,372 = 34,470, but Fig. 1 (checked as an image from PMC) shows 34,770 adults; the later steps are consistent (34,770 - 4,863 = 29,907; 29,907 - 6,991 = 22,916). The '59,842 adults' are all participants of all ages (sum of DEMO_E to DEMO_J). Results text '1486 (2.8%) for MUI' among those sitting < 7 h is inconsistent with the other percentages in the same sentence (1898 at 12.5% implies a denominator near 15,200, so 1486 would be about 9.8%). Table 1 and the 'weighted proportion' figures are unweighted shares despite their labels. Crude-model ORs in Table 2 are close to 1.0 with narrow CIs (e.g. male crude total UI 1.0 (0.997,1.0)), unlike the adjusted models; how the crude model was fitted is unclear. Software: R 4.1 and EmpowerStats; the acknowledgements thank the author of the nhanesR package. 2021-2023: KIQ042 and KIQ044 are self-administered in the MEC (ACASI) and CDC says to use the examination weight; PAD680 range 0 to 1380 minutes, 7777 refused, 9999 don't know; CDC notes that probes on PAD680 changed across cycles (a probe for times under 8 hours was added midway through 2011-2012).",
  "adjudication": null
 },
 {
  "id": "row295",
  "rank": 74,
  "row": 295,
  "doi": "10.1186/s12889-024-19722-0",
  "pmcid": "PMC11337595",
  "title": "Association of the visceral fat metabolic score with osteoarthritis risk: a cross-sectional study from NHANES 2009–2018",
  "authors": [
   "Xue, Hongfei",
   "Zhang, Longyao",
   "Xu, Jiankang",
   "Gao, Kuiliang",
   "Zhang, Chao",
   "Jiang, Lingling",
   "Lv, Sirui",
   "Zhang, Chao"
  ],
  "year": 2024,
  "journal": "BMC Public Health",
  "table_a": {
   "predictor": "Visceral fat metabolic score",
   "condition": "Osteoarthritis",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": null,
   "table_location": "Table 2, Model 3, row Q4",
   "table_quote": "| METS-VF | 1.35(1.28-1.42) | 1.23（1.15-1.31） | 1.23（1.14-1.33） |\n| p-value | p<0.01 | p<0.01 | p<0.01 |",
   "measure": "OR",
   "estimate": 2.33,
   "ci_low": 1.65,
   "ci_high": 3.28,
   "p_value": "p<0.01 (Supplementary Table 2); (p < 0.01) in Results",
   "exposure_contrast": "quartile 4 vs quartile 1 of METS-VF",
   "model_label": "Model 3",
   "covariates_in_this_model": [
    "age",
    "gender",
    "race",
    "education level",
    "poverty-to-income ratio",
    "alcohol intake",
    "smoking status",
    "work intensity",
    "recreational intensity",
    "serum calcium level",
    "vitamin D level",
    "hypertension",
    "diabetes",
    "coronary heart disease",
    "energy"
   ],
   "n_analytic": 7639,
   "n_quote": "[Abstract] This study involved 7639 participants. [Data sources and study population] Finally, 7639 participants were enrolled in this study for final analysis (Fig. 1). [comment: the number of participants per model is not printed.]",
   "events": 937
  },
  "cycles": [
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=20",
   "inclusion": "NHANES 2009-2018 participants aged 20 or older with METS-VF, osteoarthritis data, plausible dietary energy, and complete listed covariates",
   "exclusions": [
    "missing METS-VF: n = 35301 (Fig. 1: 14,392 remain)",
    "missing OA: n = 4103 (Fig. 1: 10,289 remain)",
    "age < 20 years: n = 233",
    "total energy deficit: n = 558",
    "total energy intake < 500 or > 5,000 kcal/day in females, < 500 or > 8,000 kcal/day in males: n = 57",
    "missing level of education: n = 6",
    "missing poverty income ratio: n = 834",
    "missing smoking: n = 10",
    "missing alcohol consumption: n = 650",
    "missing hypertension: n = 13",
    "missing diabetes: n = 224",
    "missing coronary heart disease: n = 17",
    "missing serum calcium: n = 48 (the age, energy and covariate exclusions together are the 2,650 'without Covariates data' in Fig. 1; 7,639 remain)"
   ],
   "quote": "[Data sources and study population] A total of 49,693 individuals participated in this survey, and participants with the following missing information were excluded, including METS-VF (n = 35301), OA (n = 4103), age < 20 years (n = 233), total energy deficit (n = 558), and total energy intake extremes < 500 or > 5,000 kcal/day in females, and < 500 or > 8,000 kcal/day in male individuals (n = 57), level of education (n = 6), poverty income ratio (n = 834), smoking (n = 10), alcohol consumption (n = 650), hypertension (n = 13), diabetes (n = 224), coronary heart disease (n = 17), and serum calcium (n = 48). Finally, 7639 participants were enrolled in this study for final analysis (Fig. 1). [Fig. 1 (image), box text: The number of participants in NHANES 2009-2018 (n=49693); excluded participants without METS-VF data (n=35301); Remaining participants (n=14392); excluded participants without osteoarthritis data (n=4103); Remaining participants (n=10289); excluded participants without Covariates data (n=2650); Final sample(n=7639)]"
  },
  "exposure": {
   "definition": "METS-VF = 4.466 + 0.011 x (ln METS-IR)^3 + 3.239 x (ln WHtR)^3 + 0.319 x sex (male = 1, female = 0) + 0.594 x ln(age); METS-IR = ln(2 x fasting glucose + fasting triglycerides) x BMI / ln(HDL-C); WHtR = waist circumference / height. The cubes are superscripts in the source ('3'). Units for glucose, triglycerides and HDL-C are not stated (the original METS-IR uses mg/dL).",
   "nhanes_variables": [
    "LBXGLU",
    "LBXTR (LBXTLG in 2021-2023)",
    "LBDHDD",
    "BMXBMI",
    "BMXWAIST",
    "BMXHT",
    "RIAGENDR",
    "RIDAGEYR"
   ],
   "nhanes_files": [
    "GLU",
    "TRIGLY",
    "HDL",
    "BMX",
    "DEMO"
   ],
   "transform": "none (per 1 unit) in the headline model; quartiles in Table 2",
   "categories": "[comment: quartiles Q1 to Q4 with Q1 as reference in Table 2; cutpoints not reported.] [Table 2, header] | Model | Quartiles of METS-VF <colspan=4> | p for trend |",
   "quote": "[Assessment of visceral fat metabolic score] The METS-VF is an index which can be adopted for assessing the visceral fat accumulation and associated metabolic health of an individual. In this study, METS-VF was calculated using the following formula: METS-VF = 4.466 + 0.011[(Ln (METS-IR))3] + 3.239[(Ln (WHtR))3] + 0.319(Sex) + 0.594(Ln (Age)) (“male” = 1, “female” = 0). The metabolic insulin resistance score (METS-IR) was calculated with the formula: METS-IR = Ln [(2 × fasting glucose) + fasting triglycerides) × BMI] / [Ln (high-density lipoprotein cholesterol)]. In addition, waist-to-height ratio (WHtR) was calculated by WHtR = WC / HT."
  },
  "outcome": {
   "definition": "Self-reported doctor-diagnosed arthritis (MCQ160A = yes) with the type reported as osteoarthritis (2009-2010: MCQ191 code 2 'Osteoarthritis'; 2011-2018: MCQ195 code 1 'Osteoarthritis or degenerative arthritis'; 2021-2023: MCQ195 code 1, same label). Whether participants reporting other arthritis types (rheumatoid, psoriatic, other, refused or don't know) were counted as non-OA or excluded is not stated.",
   "nhanes_variables": [
    "MCQ160A",
    "MCQ191 (2009-2010)",
    "MCQ195 (2011-2018 and 2021-2023)"
   ],
   "nhanes_files": [
    "MCQ"
   ],
   "quote": "[Assessment of osteoarthritis] OA was assessed using a NHANES codebook questionnaire in the form of “Has a doctor or other health professional ever told you that you have arthritis?“. The response options were “yes” or “no”. Those who chose “yes” went on to the next round of the questionnaire with the question “What type of arthritis is this?” Those who selected the option of OA were included in the study."
  },
  "covariates": [
   {
    "name": "age",
    "coding": "unstated (Table 1 reports survey-weighted mean (SD))",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Also a component of METS-VF."
   },
   {
    "name": "gender",
    "coding": "male, female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Also a component of METS-VF."
   },
   {
    "name": "race",
    "coding": "Mexican American, Other Hispanic, Non-Hispanic White, Non-Hispanic Black, Other Race (Table 1)",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Variable not named; the five groups match RIDRETH1."
   },
   {
    "name": "education level",
    "coding": "Below high school, High school, Above high school (Table 1)",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Mapping of DMDEDUC2 codes unstated."
   },
   {
    "name": "poverty-to-income ratio",
    "coding": "< 1.3, 1.3–3.5, > 3.5 (Table 1)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "alcohol intake",
    "coding": "whether they consume at least 12 alcoholic beverages per year (Table 1: Yes, No)",
    "nhanes_variables": [
     "ALQ101 (2009-2016)"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": false,
    "note": "The 'at least 12 drinks in any one year' item (ALQ101) is absent from 2017-2018 (ALQ_J) and 2021-2023 (ALQ_L); how the paper coded 2017-2018 is unstated. ALQ_L has ALQ111 (ever had a drink), ALQ121 (past-12-month frequency) and ALQ130 (drinks per drinking day), which allow only an approximation."
   },
   {
    "name": "smoking status",
    "coding": "now, former, never, from 'smoking at least 100 cigarettes in your life' and whether smoking now",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "work intensity",
    "coding": "Vigorous, Moderate, Other (Table 1 'Work activity')",
    "nhanes_variables": [
     "PAQ605",
     "PAQ620"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": false,
    "note": "PAQ_L (2021-2023) has no work-activity questions."
   },
   {
    "name": "recreational intensity",
    "coding": "Vigorous, Moderate, Other (Table 1 'Recreational activities')",
    "nhanes_variables": [
     "PAQ650",
     "PAQ665 (2009-2018)",
     "PAD810Q/PAD810U",
     "PAD790Q/PAD790U (2021-2023)"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": true,
    "note": "2021-2023 asks how often vigorous and moderate leisure-time activity is done instead of the yes/no items; 'any' can be derived as frequency > 0."
   },
   {
    "name": "serum calcium level",
    "coding": "mg/dL, continuous presumably (Table 1 mean (SD))",
    "nhanes_variables": [
     "LBXSCA"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "vitamin D level",
    "coding": "nmol/L, continuous presumably (Table 1 mean (SD))",
    "nhanes_variables": [
     "LBXVIDMS"
    ],
    "nhanes_files": [
     "VID"
    ],
    "in_2021_2023": true,
    "note": "Missing vitamin D is not among the listed exclusions."
   },
   {
    "name": "hypertension",
    "coding": "yes/no; definition unstated",
    "nhanes_variables": [
     "BPQ020"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Variable not named."
   },
   {
    "name": "diabetes",
    "coding": "yes/no; definition unstated",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "Variable not named."
   },
   {
    "name": "coronary heart disease",
    "coding": "yes/no; definition unstated",
    "nhanes_variables": [
     "MCQ160C"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "MCQ160c in MCQ_L."
   },
   {
    "name": "energy",
    "coding": "mean of the two 24-hour recalls, kcal/day",
    "nhanes_variables": [
     "DR1TKCAL",
     "DR2TKCAL"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "The covariates paragraph also lists fat, protein, carbohydrate, saturated, monounsaturated and polyunsaturated fat intake, but the Model 3 footnotes list only energy."
   }
  ],
  "design": {
   "weights": "unstated which 2-year weight (fasting subsample, MEC, or dietary two-day); the paper says the 2-year weights were divided by 2 (five cycles were pooled; dividing every weight by the same constant does not change ORs or their design-based CIs)",
   "strata_psu": "stated that all analyses account for the survey design; strata and PSU variables not named",
   "quote": "[Statistical analysis] Considering the complex sampling design and ensuring the nationally representative estimates, all analyses were adjusted for the survey design and weighting variables, for which a simple linear scaling of the 2-year weights (the original 2-year sample weights divided by 2) was performed.",
   "software": "Stata 17.0 and R (version 4.3.1)",
   "missing_data": "complete case (participants missing METS-VF, OA, or the listed covariates were excluded)",
   "quote_missing": "[Data sources and study population] participants with the following missing information were excluded, including METS-VF (n = 35301), OA (n = 4103), age < 20 years (n = 233), total energy deficit (n = 558)"
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "[Statistical analysis] Besides, the association of OA with quartiles of the METS-VF index was examined using the multivariable-adjusted logistic regression models. According to the guidelines [16], three models were developed to explore the association between METS-VF and OA. Model 1 was unadjusted for variables, Model 2 was adjusted for age and sex, while Model 3 was adjusted for all the covariates. [Results] Meanwhile, the METS-VF index was also transformed from a categorical variable to a continuous variable and incorporated into three models for weighted multivariate logistic regression analysis. The final results were consistent with those mentioned above. As revealed by the findings of model 3, the risk of developing OA increased by 23% for each unit increase in the METS-VF index (OR = 1.23, 95% CI: 1.14−1.33) (p < 0.01) (Supplementary Table 2)."
  },
  "unstated": [
   "Which 2-year weight (fasting subsample WTSAF2YR, MEC WTMEC2YR, or dietary two-day WTDR2D) and the strata and PSU variables",
   "Units of glucose, triglycerides and HDL-C in METS-IR (mg/dL in the original derivation) and of waist and height (cm presumably)",
   "Whether participants reporting arthritis of another type (rheumatoid, psoriatic, other) or of unknown type were coded as non-OA or excluded",
   "How 'alcohol intake (at least 12 drinks per year)' was coded in 2017-2018, where that question was not asked",
   "Definitions of hypertension, diabetes and coronary heart disease",
   "Coding of work and recreational intensity into Vigorous/Moderate/Other",
   "How age, serum calcium, vitamin D and energy entered the model (continuous presumably)",
   "What 'total energy deficit' means (missing recall, one recall only, or unreliable recall) and whether both recalls were required",
   "How participants missing vitamin D, work activity or recreational activity were handled (not among the listed exclusions)",
   "METS-VF quartile cutpoints",
   "Whether pregnant women were excluded",
   "Reference categories for categorical covariates",
   "Number of participants in each model"
  ],
  "notes": "Headline choice: the abstract has no effect estimate with a CI, and main-text Table 2 reports quartiles only, but the paper also fits METS-VF as a continuous variable in all three models (Supplementary Table 2, quoted in Results), so the brief's rule picks the most-adjusted continuous estimate: Model 3, OR 1.23 (1.14-1.33) per unit. If 'main regression table' is read strictly as Table 2, the alternative headline is Q4 vs Q1 in Model 3, OR 2.33 (1.65-3.28). Methods also say 'the four categorical variables of METS-VF were transformed into continuous variables' for the trend test; the Results and Supplementary Table 2 describe the continuous estimate as 'for each unit increase in the METS-VF index'. Scale warning: with the printed formula and glucose, triglycerides and HDL in mg/dL, worked examples give METS-VF of about 5.7 to 7.8 for typical adults (e.g. a 50-year-old man, BMI 28, glucose 100, triglycerides 150, HDL 45 mg/dL, waist 100 cm, height 175 cm gives 7.13), yet Table 1 reports means of 9.54 (SD 2.01) and 10.21 (SD 1.66) and the ROC cut-off is 9.552; the authors' computation or units may differ from the printed formula, which changes what 'per unit' means and could make the per-unit OR hard to reproduce. Results text misreports Q2 as 'OR = 1.14, 95% CI: 1.02−2.02' while Table 2 has 1.43 (1.02−2.01). Weights are said to be divided by 2 although five cycles are pooled; a common rescaling does not affect ORs or CIs. Age and sex enter both METS-VF and the covariate set. The fasting-subsample laboratory values (glucose, triglycerides) explain the 35,301 exclusions for missing METS-VF. Fig. 1 (checked as an image from PMC) matches the text: 49,693 to 14,392 to 10,289 to 7,639, with 2,650 excluded for covariates (the sum of the age, energy and covariate exclusions). Arthritis type items verified in CDC documentation: MCQ191 in 2009-2010 (1 Rheumatoid arthritis, 2 Osteoarthritis, 3 Psoriatic arthritis, 4 Other) and MCQ195 in 2011-2018 and 2021-2023 (1 Osteoarthritis or degenerative arthritis, 2 Rheumatoid arthritis, 3 Psoriatic arthritis, 4 Other). Table 1 labels its groups 'formers' and its smoking percentages look implausible (non-OA Never 23.89%); Supplementary Table 1 has a duplicated 'Non-Hispanic Black' row and malformed CIs; none of this affects the headline. Software: Stata 17.0 and R 4.3.1.",
  "adjudication": "The abstract reports no estimate with a 95% CI, and the paper's main regression table (Table 2) analyzes METS-VF only in quartiles; the per-unit estimate the extraction recorded is in Supplementary Table 2, which the headline rule does not use. The headline is Table 2's most-adjusted estimate for the highest quartile against the lowest."
 },
 {
  "id": "row111",
  "rank": 89,
  "row": 111,
  "doi": "10.1093/ajh/hpw010",
  "pmcid": "PMC5006109",
  "title": "Does Age Matter? Association Between Usual Source of Care and Hypertension Control in the US Population: Data From NHANES 2007–2012",
  "authors": [
   "Dinkler, John M",
   "Sugar, Catherine A",
   "Escarce, José J",
   "Ong, Michael K",
   "Mangione, Carol M"
  ],
  "year": 2016,
  "journal": "American Journal of Hypertension",
  "table_a": {
   "predictor": "Usual source of care",
   "condition": "Blood pressure",
   "population": "US population"
  },
  "headline": {
   "abstract_quote": "In adjusted analyses, those with a USOC had higher odds of hypertension control (odds ratio = 3.89, 95% confidence interval (CI): 2.15–6.98).",
   "table_location": "Table 2 (Odds ratios and 95% confidence intervals for hypertension control), column 'Model 1a' (footnote a: 'Logistic model controlling for demographics and clinical characteristics.'; the model without imputation), row 'Usual source of care'",
   "table_quote": "| Usual source of care | 3.89 (2.15–6.98) | 3.89 (2.60–5.83) |",
   "measure": "OR",
   "estimate": 3.89,
   "ci_low": 2.15,
   "ci_high": 6.98,
   "p_value": "not printed (95% CI excludes 1)",
   "exposure_contrast": "having a usual source of care (a place usually gone to when sick or needing health advice, other than the emergency department) vs no usual source of care (no such place, or the emergency department as the usual place)",
   "model_label": "Model 1 (Table 2 'Model 1a': logistic model controlling for demographics and clinical characteristics, without imputation)",
   "covariates_in_this_model": [
    "Age strata",
    "Race/ethnicity",
    "Male",
    "Married",
    "Insured",
    "Comorbid conditions: Heart failure",
    "Comorbid conditions: Diabetes",
    "Comorbid conditions: Hyperlipidemia",
    "BMI category",
    "plus covariates not shown: Table 2 note 'Not all variables in model shown in table.' (which ones is not stated)"
   ],
   "n_analytic": null,
   "n_quote": "Not reported for Model 1. Model 1 is the complete-case model (Multivariable models: \"In the full model adjusting for demographics and comorbidities without imputation\"), and Missing data (Results): \"Only 3 variables were missing approximately 10% (BMI category, income, and hyperlipidemia).\" The full hypertensive subsample, used by the imputation model, is Table 1 note \"Total sample size n = 7,653 representing 87,298,349 individuals.\" with header \"| USOC (n = 6,785)Weighted % = 90.7% | No USOC (n = 868)Weighted % = 9.3% |\"",
   "events": null
  },
  "cycles": [
   "2007-2008",
   "2009-2010",
   "2011-2012"
  ],
  "population": {
   "age": ">=18",
   "defining": "Adults with hypertension: currently taking blood pressure medication, or SBP 140 or DBP 90 at the MEC exam (the comparison operator is missing in the text; >= 140 / >= 90 presumed, the complement of the control definition)",
   "inclusion": "NHANES 2007-2012 participants aged 18 and older in the hypertensive subsample (n = 7,653)",
   "exclusions": [
    "Under 18 years (count not given)",
    "Not hypertensive: not taking blood pressure medication and SBP/DBP below 140/90 at the MEC exam (count not given)",
    "Model 1 only: participants with missing values on model variables (complete case; count not given)"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "Data source: \"The NHANES sample for this study is restricted to the hypertensive population (i.e., those who currently taking blood pressure medication or had systolic blood pressure 140 or diastolic 90 at the time of the mobile exam component).\" Measures: \"We include only adults 18 years of age and older and stratify by 10-year intervals to create 6 separate groups.\" \"Treatment for hypertension was defined by one’s response to: “Are you taking blood pressure medication?”\""
  },
  "exposure": {
   "definition": "Usual source of care (USOC): a place usually gone to when ill and needing care. Respondents reporting no place, or the emergency department as the place, are coded 'no USOC'; everyone else with a place (clinic or doctor's office, hospital outpatient department) is 'USOC'.",
   "nhanes_variables": [
    "HUQ030",
    "HUQ040"
   ],
   "nhanes_files": [
    "HUQ"
   ],
   "transform": "binary: USOC vs no USOC (none or emergency department)",
   "categories": "Binary. Measures: \"If individuals report no USOC or use the emergency department, they are treated as “no USOC.”\" (variable names not given in the paper; HUQ030 and HUQ040 are the 2007-2012 items, HUQ030 and HUQ042 in 2021-2023)",
   "quote": "Measures: \"NHANES defines USOC as a place to go when one is ill and needs care; places are listed as hospital outpatient department, outpatient clinic or doctor’s office, emergency department, or none.14 If individuals report no USOC or use the emergency department, they are treated as “no USOC.” For some analyses, USOC type is broken down into “traditional” USOC (clinic, doctor’s office, or outpatient hospital department), emergency department USOC, and no USOC.\""
  },
  "outcome": {
   "definition": "Hypertension control: average SBP < 140 mmHg and average DBP < 90 mmHg (three readings, averaged after discarding the first), among hypertensive adults",
   "nhanes_variables": [
    "BPXSY1",
    "BPXSY2",
    "BPXSY3",
    "BPXDI1",
    "BPXDI2",
    "BPXDI3"
   ],
   "nhanes_files": [
    "BPX"
   ],
   "quote": "Measures: \"Trained professionals measured blood pressure using sphygmomanometry and appropriately sized arm cuffs after 5 minutes seated rest. Blood pressure measurements were taken 3 consecutive times and averaged after discarding the first measurement.13\" \"“Hypertension control” is defined as average systolic blood pressure less than 140mm Hg and diastolic less than 90mm Hg.2\""
  },
  "covariates": [
   {
    "name": "age strata",
    "coding": "6 groups by 10-year intervals: 18–34, 35–44, 45–54, 55–64, 65–74 (reference), >74 (Table 2; Table 1 labels the top group '>75')",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Model 1 (Table 2)."
   },
   {
    "name": "race/ethnicity",
    "coding": "Hispanic, non-Hispanic White (reference), non-Hispanic Black, other race including multiracial",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Model 1 (Table 2). Hispanic presumably combines Mexican American and Other Hispanic (not stated)."
   },
   {
    "name": "gender",
    "coding": "male vs female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Model 1 (Table 2 row 'Male')."
   },
   {
    "name": "marital status",
    "coding": "married or living with partner ('married') vs all others",
    "nhanes_variables": [
     "DMDMARTL"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Model 1 (Table 2 row 'Married'). DMDMARTZ in 2021-2023. Asked of ages 20+ in both eras, so 18-19-year-olds have no value; handling not stated."
   },
   {
    "name": "insurance status",
    "coding": "insured vs uninsured",
    "nhanes_variables": [
     "HIQ011"
    ],
    "nhanes_files": [
     "HIQ"
    ],
    "in_2021_2023": true,
    "note": "In Model 1 (Table 2 row 'Insured')."
   },
   {
    "name": "heart failure",
    "coding": "doctor ever told (yes/no)",
    "nhanes_variables": [
     "MCQ160B"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "In Model 1 (Table 2)."
   },
   {
    "name": "diabetes",
    "coding": "doctor ever told (yes/no)",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "In Model 1 (Table 2). Coding of 'borderline' (DIQ010 = 3) not stated."
   },
   {
    "name": "hyperlipidemia (high cholesterol)",
    "coding": "doctor ever told (yes/no)",
    "nhanes_variables": [
     "BPQ080"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "In Model 1 (Table 2). About 10% missing (Results)."
   },
   {
    "name": "BMI category",
    "coding": "underweight < 18.5, normal 18.5–24.9 (reference), overweight 25–29.9, obese >= 30 kg/m2",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "In Model 1 (Table 2). About 10% missing (Results)."
   },
   {
    "name": "education",
    "coding": "3 categories: did not complete high school; high school graduate (Table 1: 'High school graduate or some college'); college graduate",
    "nhanes_variables": [
     "DMDEDUC2",
     "DMDEDUC3"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Inclusion in Model 1 not stated. DMDEDUC2 covers 20+ only; 18-19-year-olds had DMDEDUC3 in 2007-2012, which DEMO_L lacks."
   },
   {
    "name": "income",
    "coding": "family income to poverty ratio in 4 categories: <150%, 150–249%, 250–349%, >350% FPL",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Inclusion in Model 1 not stated. About 10% missing (Results)."
   },
   {
    "name": "self-reported health",
    "coding": "fair, poor, good, very good, or excellent",
    "nhanes_variables": [
     "HUQ010"
    ],
    "nhanes_files": [
     "HUQ"
    ],
    "in_2021_2023": true,
    "note": "Inclusion in Model 1 not stated."
   },
   {
    "name": "physical activity",
    "coding": "met the American College of Sports Medicine guidelines for aerobic physical activity (yes/no)",
    "nhanes_variables": [
     "PAQ605",
     "PAQ620",
     "PAQ635",
     "PAQ650",
     "PAQ665"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": false,
    "note": "Inclusion in Model 1 not stated; items and thresholds not stated. The 2007-2012 questionnaire covered work, transport and leisure; PAQ_L in 2021-2023 has leisure-time moderate and vigorous activity only (PAD790Q/U, PAD800, PAD810Q/U, PAD820)."
   },
   {
    "name": "smoking",
    "coding": "at least 100 cigarettes in lifetime; Table 1 shows nonsmoker, current smoker, former smoker",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "Inclusion in Model 1 not stated."
   },
   {
    "name": "prior heart attack (myocardial infarction)",
    "coding": "doctor ever told (yes/no)",
    "nhanes_variables": [
     "MCQ160E"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "Inclusion in Model 1 not stated."
   },
   {
    "name": "prior stroke",
    "coding": "doctor ever told (yes/no)",
    "nhanes_variables": [
     "MCQ160F"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "Inclusion in Model 1 not stated."
   },
   {
    "name": "COPD",
    "coding": "Table 1 only; definition not stated",
    "nhanes_variables": [
     "MCQ160G",
     "MCQ160K"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "Inclusion in Model 1 not stated. 2007-2012 asked emphysema (MCQ160G) and chronic bronchitis (MCQ160K) separately; 2021-2023 has one combined item MCQ160p (COPD, emphysema, or chronic bronchitis)."
   },
   {
    "name": "kidney function",
    "coding": "serum creatinine (mg/dl), continuous",
    "nhanes_variables": [
     "LBXSCR"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Inclusion in Model 1 not stated."
   }
  ],
  "design": {
   "weights": "Survey-weighted, but the weight is not named. Blood pressure comes from the MEC exam, so a 6-year MEC weight (WTMEC2YR/3) is the usual choice; the paper does not say.",
   "strata_psu": "Not stated (only 'survey methods')",
   "quote": "Study design and statistical methods: \"We use survey methods for all weighted bivariate analyses and regression models.\" \"Predictions are generated using average probabilities among actual persons in the data and errors are weighted to account for population sampling.\"",
   "software": "not stated",
   "missing_data": "complete case for the headline Model 1 ('without imputation'); Model 2 uses multiple imputation by chained equations (5 imputed datasets; imputation variables age, gender, race/ethnicity, diabetes status, smoking status, USOC)",
   "quote_missing": "Missing data (Methods): \"Variables with missing data were imputed using multiple imputation with chained equations using age, gender, race/ethnicity, diabetes status, smoking status, and USOC.18–20 We specified 5 multiply imputed datasets and variables were assumed to be missing at random.21\" Missing data (Results): \"Most of the variables had either no missing data or were missing <2%. Only 3 variables were missing approximately 10% (BMI category, income, and hyperlipidemia). All missing data were imputed as described above.\""
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "Statistical methods: \"We employed 2 logistic models to analyze the effects of USOC on hypertension control.\" Abstract: \"Multivariable logistic regression was used to evaluate the association between having a USOC and hypertension control.\""
  },
  "unstated": [
   "Which survey weight (MEC vs interview; how the three 2-year weights were combined) and whether strata and PSUs were used",
   "The full covariate list of Model 1: Table 2 shows age strata, race/ethnicity, male, married, insured, heart failure, diabetes, hyperlipidemia and BMI category and says not all variables are shown; whether education, income, self-reported health, physical activity, smoking, prior MI, prior stroke, COPD and creatinine were in the model is not stated",
   "The analytic N and number with controlled hypertension in the complete-case Model 1",
   "How HUQ030 = 3 ('more than one place'), HUQ040 = 5 ('some other place'), and refused/don't know answers were classified",
   "Comparison operators of the hypertension inclusion criterion (text: 'systolic blood pressure 140 or diastolic 90'); presumably >= 140 or >= 90",
   "Which readings were averaged for the inclusion criterion, and handling of participants with fewer than three readings (2007-2012 could have a fourth reading)",
   "How 'currently taking blood pressure medication' was derived (BPQ050A is asked only of those told to take medicine, BPQ040A = 1)",
   "Coding of education and marital status for 18-19-year-olds (DMDEDUC2 and DMDMARTL target 20+)",
   "Income category boundaries ('250–349%' and '>350%' leave 3.50 ambiguous)",
   "Derivation of meeting the ACSM physical activity guidelines (items, domains, thresholds)",
   "Diabetes coding for 'borderline' answers; COPD definition",
   "Handling of pregnant women (not mentioned)",
   "Software (the marginal-effects approach is typical of Stata 'margins' but no software is named)"
  ],
  "notes": "Headline is Table 2 Model 1 (complete case), the only OR the abstract reports; Model 2 (MICE, 5 imputations) gives the same OR with CI 2.60-5.83. The study population is hypertensive adults (treated or with SBP/DBP at or above 140/90), narrower than Suchak et al.'s 'US population' label; outcome is the binary 'hypertension controlled' (< 140/90). The emergency department as the usual place counts as 'no USOC'. Table 1 labels the oldest age group '>75' while Table 2 and the text use '>74' (reference group 65–74). The 2007-2012 BP was auscultatory; 2021-2023 has only oscillometric readings (BPXO_L). The 2021-2023 type-of-place item (HUQ042) adds urgent care/retail clinic and VA categories; classifying them is a protocol decision. Household interviews in 2021-2023 could be by telephone. Source text is the PMC HTML; no erratum or retraction found. Supplements: none (Europe PMC hasSuppl = N). Variable names are not given in the paper; those listed are the NHANES items that match the descriptions.",
  "adjudication": null
 },
 {
  "id": "row014",
  "rank": 90,
  "row": 14,
  "doi": "10.3389/fnut.2023.1016809",
  "pmcid": "PMC10011108",
  "title": "Association between serum 25-hydroxyvitamin D and osteoarthritis: A national population-based analysis of NHANES 2001–2018",
  "authors": [
   "Yu, Guoyu",
   "Lin, Yuan",
   "Dai, Hanhao",
   "Xu, Jie",
   "Liu, Jun"
  ],
  "year": 2023,
  "journal": "Frontiers in Nutrition",
  "table_a": {
   "predictor": "Serum vitamin D concentrations",
   "condition": "Osteoarthritis",
   "population": "US population"
  },
  "headline": {
   "abstract_quote": "Higher serum 25(OH)D levels were associated with more osteoarthritis prevalence in fully adjusted model (odd ratio [OR] 1.25 [95% CI: 1.10, 1.43] for the 50–75 nmol/L group; OR 1.62 [95% CI: 1.42, 1.85] for the 75–100 nmol/L group; OR 1.91 [95% CI: 1.59, 2.30] for the ≥100 nmol/L group; with <50 nmol/L group as the reference) (p < 0.001 for trend).",
   "table_location": "Table 3, fully adjusted model, row >= 100 nmol/L",
   "table_quote": "| ≥100 nmol/L | 2.56 (2.16, 3.04)1 | 1.93 (1.61, 2.32)1 | 1.86 (1.55, 2.25)1 | 1.91 (1.59, 2.30)1 |",
   "measure": "OR",
   "estimate": 1.91,
   "ci_low": 1.59,
   "ci_high": 2.3,
   "p_value": "< 0.001 for trend",
   "exposure_contrast": "serum 25(OH)D >= 100 nmol/L vs < 50 nmol/L (reference)",
   "model_label": "fully adjusted model",
   "covariates_in_this_model": [
    "age",
    "gender",
    "race",
    "education level",
    "PIR",
    "BMI",
    "season of examination",
    "alcohol consumption",
    "smoking status",
    "recreational physical activity",
    "vitamin D supplements",
    "self-reported health"
   ],
   "n_analytic": 21334,
   "n_quote": "Abstract: \"Among the 21,334 participants included (weighted mean age, 56.9 years; 48.5% men)\"; Study population: \"Finally, this study included a large national representative sample (n = 21,334) of the general adult US population.\"",
   "events": null
  },
  "cycles": [
   "2001-2002",
   "2003-2004",
   "2005-2006",
   "2007-2008",
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=40",
   "defining": "US adults aged 40 and older",
   "inclusion": "NHANES 2001-2018 participants aged 40+ with serum 25(OH)D, osteoarthritis information and all covariates",
   "exclusions": [
    "Under 40 years (n = 58,284)",
    "Missing serum 25(OH)D (n = 3,702)",
    "Missing osteoarthritis information (n = 3,641)",
    "Missing other covariates (n = 4,390)"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "Study population: \"Participants being under 40 years old (n = 58,284) were excluded. We further excluded participants with missing data on serum 25(OH)D concentrations (n = 3,702), osteoarthritis information (n = 3,641), and other covariates (n = 4,390). Finally, this study included a large national representative sample (n = 21,334) of the general adult US population.\""
  },
  "exposure": {
   "definition": "Total serum 25(OH)D (25(OH)D3 + 25(OH)D2), nmol/L; RIA in 2001-2006 converted by CDC regression to LC-MS/MS equivalents, LC-MS/MS in 2007-2018",
   "nhanes_variables": [
    "LBDVIDMS",
    "LBXVIDMS"
   ],
   "nhanes_files": [
    "VID"
   ],
   "transform": "categories: clinical cut-offs <50 (reference), 50–75, 75–100, >=100 nmol/L (also per 10 nmol/L increase, continuous)",
   "categories": "<50 (reference), 50–75, 75–100, >=100 nmol/L. Table 3 rows \"| <50 nmol/L | 1 (reference) | 1 (reference) | 1 (reference) | 1 (reference) |\" and Table 1 header \"| Characteristics | Total (n = 21,334) | <50 (n = 6,453) | 50–75 (n = 8,094) | 75–100 (n = 4,759) | ≥100 (n = 2,028) |\". Statistical analysis lists the groups with a typo (75–100 printed as a second 50–75): \"(<50 nmol/L; 50–75 nmol/L; 50–75 nmol/L; ≥100 nmol/L)\". Which category takes exactly 75 or 100 is not stated beyond '≥100'.",
   "quote": "2.2: \"Total serum 25(OH)D, calculated as the sum of 25(OH)D3 and 25(OH)D2, is the best indicator of vitamin D levels. Serum 25(OH)D concentrations were tested using a radioimmunoassay kit (DiaSorin, Stillwater, MN, USA) in NHANES 2001–2006. The CDC applied a more analytically accurate assay involving ultra-high performance liquid chromatography-tandem mass spectrometry (LC-MS/MS) method in NHANES 2007–2018. To compare 25(OH)D levels across cycles, the CDC standardized 25(OH)D concentrations measured from radioimmunoassays to predicted LC-MS/MS equivalents using regression equations (15).\" 2.5: \"Odds ratio (OR) with a corresponding 95% confidence interval (CI) was calculated using 25(OH)D concentrations both as continuous (per 10 nmol/L increase) and as categorical variables.\""
  },
  "outcome": {
   "definition": "Self-reported osteoarthritis: 'yes' to ever told by a doctor or other health professional of arthritis, and type 'OA' (2001-2010) or 'OA or degenerative arthritis' (2011-2018); all others (no arthritis, presumably also other arthritis types) are non-OA",
   "nhanes_variables": [
    "MCQ160A",
    "MCQ190",
    "MCQ191",
    "MCQ195"
   ],
   "nhanes_files": [
    "MCQ"
   ],
   "quote": "2.3: \"Each participant was defined as having OA if he/she responded “yes” to the question “Has a doctor or other health professional ever told you that you had arthritis?” and selected “OA” (2001–2010) or “OA or degenerative arthritis” (2011–2018) to the subsequent question “Which type of arthritis was it?”\""
  },
  "covariates": [
   {
    "name": "age",
    "coding": "years, continuous",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "gender",
    "coding": "male or female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "race",
    "coding": "4 categories: Non-Hispanic White, Non-Hispanic Black, Mexican Americans, Other Race (Other Hispanic folded into Other)",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Reference category not stated."
   },
   {
    "name": "education level",
    "coding": "3 categories: under high school, high school or equivalent, college or above (Table 1: <High school, High school, >High school)",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "PIR",
    "coding": "3 groups: low (PIR < 1.3), middle (1.3 <= PIR < 3.5), high (PIR >= 3.5)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "BMI",
    "coding": "3 groups: low to normal (< 25), overweight (25–30), obesity (>= 30 kg/m2)",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Boundary at 25 and 30 as printed ('25–30', '≥30')."
   },
   {
    "name": "season of examination",
    "coding": "November–April (winter) vs May–October (summer)",
    "nhanes_variables": [
     "RIDEXMON"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "RIDEXMON is in DEMO_L with the same two 6-month periods."
   },
   {
    "name": "alcohol consumption",
    "coding": "yes/no to 'In any one year, had at least 12 drinks of any type of alcoholic beverage?'",
    "nhanes_variables": [
     "ALQ101"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": false,
    "note": "ALQ101 ('Had at least 12 alcohol drinks/1 yr?') was asked through 2015-2016 and dropped in 2017-2018 (ALQ_J) and 2021-2023 (ALQ_L, which has ALQ111 ever had a drink, ALQ121 past-12-month frequency, ALQ130 drinks per drinking day). How the paper coded 2017-2018 is not stated."
   },
   {
    "name": "smoking status",
    "coding": "never, former, current, from ever smoking 100 cigarettes and smoking now (Table 1 labels: Never smoker, Ever smoker, Current smoker)",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "recreational physical activity",
    "coding": "active (any moderate or vigorous recreational activity over the past 30 days in 2001-2006, or in a typical week in 2007-2018) vs inactive",
    "nhanes_variables": [
     "PAD200",
     "PAD320",
     "PAQ650",
     "PAQ665"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": true,
    "note": "Item names not given (listed ones are the matching 2001-2006 and 2007-2018 items). 2021-2023 PAQ_L asks frequency of moderate and vigorous leisure-time activity (PAD790Q/U, PAD810Q/U); 'active' = any frequency > 0."
   },
   {
    "name": "vitamin D supplements",
    "coding": "vitamin D-containing supplement use in the past 30 days (text: 'more than one dietary supplement containing vitamin D'), yes/no",
    "nhanes_variables": [
     "DSDCOUNT",
     "DSQTVD"
    ],
    "nhanes_files": [
     "DSQTOT",
     "DSQIDS"
    ],
    "in_2021_2023": true,
    "note": "File and variable not named. 2021-2023 DSQIDS_L has vitamin D per product (DSQIVD) and DSQTOT_L total supplement vitamin D (DSQTVD); the 30-day supplement questions were asked by telephone after the first dietary recall in 2021-2023."
   },
   {
    "name": "self-reported health",
    "coding": "3 groups (Table 1): Fair/poor, Moderate, Excellent/very good",
    "nhanes_variables": [
     "HUQ010"
    ],
    "nhanes_files": [
     "HUQ"
    ],
    "in_2021_2023": true,
    "note": "Which HUQ010 answer is 'Moderate' is not stated (presumably 'Good')."
   }
  ],
  "design": {
   "weights": "Survey-weighted; the weight is not named: 'The new sample weight was created under the NHANES analytical guidelines.' (for nine 2-year cycles with a lab exposure, the guideline choice would be the MEC weight divided by 9; not stated)",
   "strata_psu": "Survey design accounted for with the R survey package; strata and PSU variables not named",
   "quote": "2.5: \"To ensure nationally representative estimates, analyses were adjusted for sampling weights and survey design with the R “survey” package version 4.1-1. The new sample weight was created under the NHANES analytical guidelines.\"",
   "software": "R 4.2.0, R survey package 4.1-1",
   "missing_data": "complete case (participants missing 25(OH)D, OA or any covariate excluded)",
   "quote_missing": "2.1: \"We further excluded participants with missing data on serum 25(OH)D concentrations (n = 3,702), osteoarthritis information (n = 3,641), and other covariates (n = 4,390).\""
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "Abstract (Methods): \"Multivariable logistic regression analysis was employed to assess the association between serum 25(OH)D and osteoarthritis.\" 2.5: \"analyses were adjusted for sampling weights and survey design with the R “survey” package version 4.1-1.\""
  },
  "unstated": [
   "Which weight (MEC presumably) and how the nine cycles' weights were combined; strata and PSU variable names",
   "Which category contains exactly 75 and 100 nmol/L at the boundaries other than '≥100' (the Methods list misprints 75–100 as a second 50–75)",
   "How alcohol consumption was coded in 2017-2018, when the '12 drinks in any one year' item (ALQ101) was no longer asked",
   "Coding of refused/don't know arthritis type, and whether people with other arthritis types (rheumatoid, psoriatic, other) are in the non-OA group",
   "Exact vitamin D supplement definition ('more than one dietary supplement containing vitamin D' may mean at least one) and the source variable",
   "Which HUQ010 answer was coded 'Moderate' self-reported health",
   "Reference categories of the categorical covariates",
   "Recreational physical activity items used in each cycle",
   "Whether 25(OH)D for 2001-2006 was taken from the current files (LBDVIDMS, revised October 2015) or converted by the authors with the CDC regression equations (the text says only that the CDC standardized the values)"
  ],
  "notes": "Headline follows the clarified rule: the abstract's first coding is the ordered clinical cut-offs with the lowest group as reference, represented by its highest category (>=100 vs <50 nmol/L, Model 4, 1.91 (1.59, 2.30)); the other contrasts are 1.25 (1.10, 1.43) for 50–75 and 1.62 (1.42, 1.85) for 75–100, and per 10 nmol/L is 1.07 (1.05, 1.09). The >=100 group is the smallest (n = 2,028 unweighted, Table 1 header). Models 3 and 4 adjust for vitamin D supplement use. The abstract's subgroup sentence has a typo ('BMI < 25 kg/m2, 1.01 [95% CI: 1.04, 1.08]'), the Results give 1.04 (1.00, 1.08); subgroups are not headline. The alcohol covariate ('at least 12 drinks in any one year', ALQ101) cannot be built in 2021-2023 and was not asked in 2017-2018 either. Season of examination (RIDEXMON) and the OA items exist in 2021-2023. 2021-2023 vitamin D is LC-MS/MS (same assay class as 2007-2018). Supplementary Table 1 (unweighted complete vs missing comparison) was not needed. Not a 2017-March 2020 paper.",
  "adjudication": "Under the headline rule as settled, a coding with ordered categories and the lowest as reference is represented by its highest category against the reference; the extraction took the first significant contrast the abstract listed."
 },
 {
  "id": "row327",
  "rank": 93,
  "row": 327,
  "doi": "10.3389/fnut.2024.1415484",
  "pmcid": "PMC11408230",
  "title": "Association of systemic immune biomarkers with metabolic dysfunction-associated steatotic liver disease: a cross-sectional study of NHANES 2007–2018",
  "authors": [
   "Wang, Yong",
   "Chen, Shude",
   "Tian, Chen",
   "Wang, Qi",
   "Yang, Zhihua",
   "Che, Wieqi",
   "Li, Yike",
   "Luo, Yang"
  ],
  "year": 2024,
  "journal": "Frontiers in Nutrition",
  "table_a": {
   "predictor": "Systemic immune biomarkers",
   "condition": "Metabolic-associated fatty liver conditions",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "The prevalence of MASLD increased with the Q4 of SII [OR = 1.47, 95%CI (1.24, 1.74)], SIRI [OR = 1.30, 95%CI (1.09, 1.54)], NLR [OR = 1.25, 95%CI (1.04, 1.49)], PLR [OR = 1.29, 95%CI (1.09, 1.53)] and NPAR [OR = 1.29, 95%CI (1.09, 1.54)] compared to the Q1 after adjusting for the bias caused by potential confounders.",
   "table_location": "No regression table in the main text. Main text: Figure 3 (forest-plot table, transcribed from the image), block 'SII', row 'Quartile 4', column 'Model 3'; Results: \"After fully adjusting for potential confounders, SII [OR = 1.47; 95% CI (1.24, 1.74), Q4 of SII vs. Q1] was associated with the highest OR per standard deviation increment.\" Full model output: Supplementary Table S4(A) ('The Variables in the Equation for SII and MASLD in Model 3'), row 'SII (3)' (= Q4 vs Q1; the S4(A) values match Figure 3 Model 3); supplement fetched from Europe PMC to scratchpad/dl/supp/row327/.",
   "table_quote": "Figure 3 image, SII, Quartile 4, Model 3 (transcribed): '1.47 (1.24 to 1.74)', P-value '0.000'. Supplementary Table S4(A): \"|  | SII (3) | 0.386 | 0.086 | 20.246 | 1 | 0.000 | 1.471 | 1.244 | 1.741 |\"",
   "measure": "OR",
   "estimate": 1.47,
   "ci_low": 1.24,
   "ci_high": 1.74,
   "p_value": "0.000 (Figure 3); Sig. 0.000 (Table S4(A))",
   "exposure_contrast": "SII quartile 4 (625.74 to 28397.28) vs quartile 1 (1.53 to 313.50, reference), SII = platelet x neutrophil / lymphocyte counts (each in 10^3 cells/uL); SII is the first biomarker in the abstract's list",
   "model_label": "Model 3",
   "covariates_in_this_model": [
    "gender",
    "age",
    "race/ethnicity",
    "PIR",
    "education",
    "marital status",
    "health insurance",
    "tobacco use",
    "alcohol use",
    "hypertension",
    "T2DM",
    "cardiovascular disease",
    "WC",
    "PA",
    "body mass index (BMI)",
    "TG",
    "HDL",
    "ALT",
    "AST",
    "GGT"
   ],
   "n_analytic": 14413,
   "n_quote": "Abstract: \"In total, 14,413 participants were included and 6,518 had MASLD.\" Missing covariates were multiply imputed, so the models use all 14,413 (no model N is printed).",
   "events": 6518
  },
  "cycles": [
   "2007-2008",
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=20",
   "defining": "US adults aged 20 and older",
   "inclusion": "NHANES 2007-2018 adults aged 20+ with the liver-index data and complete blood count",
   "exclusions": [
    "Age < 20 (text: n = 2,572; Figure 1 flowchart image: n = 25072; 59,842 - 34,770 = 25,072)",
    "Missing data of liver (n = 20,000)",
    "Missing data of complete blood cell (n = 57)",
    "Excluded by the MASLD definition (hepatitis B or C, excessive alcohol, drug, liver cancer, autoimmune liver disease): whether these people were dropped or counted as non-MASLD, and how many, is not stated"
   ],
   "exclusions_not_in_2021_2023": [
    "Drug-induced (secondary) liver disease ('and drug' in the MASLD definition): 2021-2023 prescription data have no drug names, and the paper does not list the drugs"
   ],
   "quote": "Results: \"Among a total of 59,842 subjects in the NHANES 2007–2018, we included 34,770 subjects aged ≥20 years (53). Of these, 45,129 subjects who met the following criteria were excluded: (1) Age < 20 (n = 2,572); (2) Missing data of liver (n = 20,000); (3) Missing data of complete blood cell (n = 57). Finally, 14,413 subjects were included in the analysis, of which 6,518 were participants with MASLD, and 7,985 were participants without MASLD (Figure 1).\" Figure 1 flowchart (image, transcribed): 'NHANES 2007-2018 (n=59842)' -> 'Age < 20 years (n=25072)' -> 'Age ≥ 20 (n=34770)' -> excluded 'Missing liver data (n=20000)', 'Missing blood cell data (n=57)' -> 'Participations included (n=14413)' -> 'Non-MASLD (n=7985)', 'MASLD (n=6518)'."
  },
  "exposure": {
   "definition": "Systemic immune-inflammation index SII = platelet count x neutrophil count / lymphocyte count, counts in 10^3 cells/uL (Beckman Coulter CBC), in quartiles with Q1 as reference",
   "nhanes_variables": [
    "LBXPLTSI",
    "LBDNENO",
    "LBDLYMNO"
   ],
   "nhanes_files": [
    "CBC"
   ],
   "transform": "categories: quartiles (Q1 reference) at the cutpoints in the Figure 3 legend",
   "categories": "Figure 3 legend: \"For SII, Median [Range]: Quartiles 1, 244.38 [1.53 to 313.50]; Quartiles 2, 374.29 [313.51 to 440.00]; Quartiles 3, 520.00 [440.01 to 625.73]; Quartiles 4, 804.15 [625.74 to 28397.28];\"",
   "quote": "Exposure variable: \"Lymphocyte, neutrophil, and platelet counts, expressed as ×103 cells/μL, were measured using automated hematology analyzing devices. The following formulas were used to calculate immune-inflammatory markers: (1) Systemic Immune-Inflammation Index (SII) = platelet count * neutrophils count/lymphocytes count;\""
  },
  "outcome": {
   "definition": "MASLD = hepatic steatosis (Fatty Liver Index >= 60 or US Fatty Liver Index >= 30) in the absence of hepatitis B (HBsAg positive) or C (HCV antibody or RNA positive), secondary liver disease from excessive alcohol (> 1 drink/day women, > 2 drinks/day men) or drugs, liver cancer, and autoimmune liver disease. FLI = e^(0.953*ln(TG) + 0.139*BMI + 0.718*ln(GGT) + 0.053*waist - 15.745) / (1 + e^(same)) * 100. USFLI = e^(0.3458*Mexican American - 0.8073*non-Hispanic black + 0.0093*age + 0.6151*ln(GGT) + 0.0249*waist + 1.1792*insulin + 0.8242*ln(glucose) - 14.7812) / (1 + e^(same)) * 100 as printed (the published USFLI, Ruhl and Everhart 2015, uses ln(insulin)). No cardiometabolic criterion is required, so this is the older NAFLD-type definition under the MASLD name.",
   "nhanes_variables": [
    "LBXTR",
    "LBXSTR",
    "BMXBMI",
    "LBXSGTSI",
    "BMXWAIST",
    "RIDRETH1",
    "RIDAGEYR",
    "LBXIN",
    "LBXGLU",
    "LBDHBG",
    "LBXHCR",
    "ALQ130",
    "MCQ230A"
   ],
   "nhanes_files": [
    "TRIGLY",
    "BIOPRO",
    "BMX",
    "DEMO",
    "INS",
    "GLU",
    "HEPBD",
    "HEPC",
    "ALQ",
    "MCQ"
   ],
   "quote": "Definition of MASLD: \"Fatty Liver Index (FLI) and United States Fatty Liver Index (USFLI) ranged from 0 to 100 using the following formula (24, 25):\" \"FLI=(e0.953∗ln(TG)+0.139∗BMI+0.718∗ln(GGT)+0.053∗waist circumference−15.745)/ (1+e0.953∗ln(TG)+0.139∗BMI+0.718∗ln(GGT)+0.053∗waist circumference−15.745)∗100;\" \"USFLI=e(0.3458∗Mexican American−0.8073∗non−Hispanic black+0.0093∗age+0.6151∗lnGGT+0.0249∗waist circumference+1.1792∗insulin+0.8242∗lnGlucose−14.7812)/ (1+e(0.3458∗Mexican American−0.8073∗non−Hispanicblack+0.0093∗age+0.6151∗lnGGT+0.0249∗waist circumference+1.1792∗insulin+0.8242∗lnGlucos−14.7812))∗100\" \"Hepatic steatosis was defined as a FLI ≥60 or a USFLI ≥30 (26). MASLD was defined as the presence of hepatic steatosis in the absence of (1) hepatitis B (positive hepatitis B surface antigen) or hepatitis C infection (positive hepatitis C antibody or HCV RNA); (2) the possibility of secondary liver disease caused by excessive alcohol consumption (more than 12 drinks in the past year, with all others considered non-drinkers; >1 alcoholic drink/day for women or >2 alcoholic drinks/day for men) and drug; (3) liver cancer; (4) autoimmune liver disease.\""
  },
  "covariates": [
   {
    "name": "gender",
    "coding": "male or female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "age",
    "coding": "years, continuous (Table S4(A) has a single 'Age' coefficient)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "race/ethnicity",
    "coding": "5 categories: Mexican American, other Hispanic, non-Hispanic White, non-Hispanic Black, other including multi-racial",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Reference category not stated (SPSS indicator coding)."
   },
   {
    "name": "education",
    "coding": "below high school vs high school above",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "PIR",
    "coding": "poor (PIR < 1.3) vs not poor (>= 1.3)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Multiply imputed when missing."
   },
   {
    "name": "health insurance",
    "coding": "yes/no",
    "nhanes_variables": [
     "HIQ011"
    ],
    "nhanes_files": [
     "HIQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "marital status",
    "coding": "never married; widowed/divorced/separated; married/living with partner (Table 1)",
    "nhanes_variables": [
     "DMDMARTL"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DMDMARTZ in 2021-2023 has the same three groups."
   },
   {
    "name": "tobacco use",
    "coding": "smoked more than 100 cigarettes in life vs not",
    "nhanes_variables": [
     "SMQ020"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "alcohol use",
    "coding": "more than 12 drinks in the past year vs not",
    "nhanes_variables": [
     "ALQ120Q",
     "ALQ121",
     "ALQ130"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "Item not named. 2021-2023 ALQ121 (past-12-month frequency) and ALQ130 (drinks per drinking day) can count drinks in the past year."
   },
   {
    "name": "hypertension",
    "coding": "self-reported high blood pressure, antihypertensive medication, or mean SBP >= 140 and/or mean DBP >= 90 mmHg",
    "nhanes_variables": [
     "BPQ020",
     "BPQ050A",
     "BPXSY1",
     "BPXDI1"
    ],
    "nhanes_files": [
     "BPQ",
     "BPX"
    ],
    "in_2021_2023": true,
    "note": "2021-2023: BPQ020, BPQ150, BPXO_L (oscillometric)."
   },
   {
    "name": "T2DM (diabetes)",
    "coding": "ADA criteria (fasting glucose >= 126 mg/dL, 2-h OGTT glucose >= 200 mg/dL, HbA1c >= 6.5%) or yes to told diabetes, taking insulin, or taking diabetes pills",
    "nhanes_variables": [
     "LBXGLU",
     "LBXGLT",
     "LBXGH",
     "DIQ010",
     "DIQ050",
     "DIQ070"
    ],
    "nhanes_files": [
     "GLU",
     "OGTT",
     "GHB",
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "No OGTT in 2021-2023 (the 2-h glucose component cannot be built); the other components can."
   },
   {
    "name": "cardiovascular disease",
    "coding": "self-report of coronary heart disease, angina, myocardial infarction, stroke, or congestive heart failure",
    "nhanes_variables": [
     "MCQ160C",
     "MCQ160D",
     "MCQ160E",
     "MCQ160F",
     "MCQ160B"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "WC",
    "coding": "binary 'WC01' in Table S4(A); Table 1 groups: <102 for male or <88 for female vs >=102 / >=88",
    "nhanes_variables": [
     "BMXWAIST"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Waist is also an input of both FLI and USFLI (the outcome)."
   },
   {
    "name": "PA",
    "coding": "binary 'PA01': insufficiently active (MVPA < 150 min/week) vs sufficiently active (>= 150), MVPA = moderate minutes x days + vigorous minutes x days",
    "nhanes_variables": [
     "PAQ605",
     "PAQ620",
     "PAQ650",
     "PAQ665"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": false,
    "note": "Which domains were summed (work, transport, recreation in the 2007-2018 questionnaire) is not stated; PAQ_L in 2021-2023 has leisure-time moderate and vigorous activity only (PAD790Q/U, PAD800, PAD810Q/U, PAD820)."
   },
   {
    "name": "body mass index (BMI)",
    "coding": "3 categories: < 25 normal/underweight, 25 to < 30 overweight, >= 30 obesity",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "BMI is also an input of FLI (the outcome); Table S4(A) BMI(2) OR = 229.707."
   },
   {
    "name": "TG",
    "coding": "binary 'TG(1)' in Table S4(A) (Table 1: TG Yes/No); threshold not stated",
    "nhanes_variables": [
     "LBXTR",
     "LBXSTR"
    ],
    "nhanes_files": [
     "TRIGLY",
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "TG is also an input of FLI (the outcome). 2021-2023: LBXTLG (fasting) or LBXSTR."
   },
   {
    "name": "HDL",
    "coding": "binary 'HDL(1)' in Table S4(A) (Table 1: HDL Yes/No); threshold not stated",
    "nhanes_variables": [
     "LBDHDD"
    ],
    "nhanes_files": [
     "HDL"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "ALT",
    "coding": "continuous (IU/L)",
    "nhanes_variables": [
     "LBXSATSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "AST",
    "coding": "continuous (IU/L)",
    "nhanes_variables": [
     "LBXSASSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "GGT",
    "coding": "continuous (IU/L)",
    "nhanes_variables": [
     "LBXSGTSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "GGT is also an input of both FLI and USFLI (the outcome)."
   }
  ],
  "design": {
   "weights": "Not mentioned anywhere (no sample weights, strata or PSUs). Table 1 reports unweighted n (%), and the supplementary tables are standard SPSS binary logistic regression output ('Variables in the Equation': B, S.E., Wald, df, Sig.), which points to unweighted models.",
   "strata_psu": "not mentioned",
   "quote": "Statistical analysis: \"All statistical analysis were performed using IBM SPSS software Version 27 (51) and R software Version 4.2.1 (52). Odds ratio (OR) with 95% Confidence Interval (CI) were calculated using logistic regression analysis.\"",
   "software": "IBM SPSS 27 and R 4.2.1",
   "missing_data": "imputation: multiple imputation by chained equations (m = 5) for PIR, education, marital status, health insurance, tobacco use, alcohol use and BMI; pooling method not stated",
   "quote_missing": "Statistical analysis: \"Missing values (PIR, education, marital status, health insurance, tobacco use, alcohol use and BMI) were imputed using the Multiple Imputation by Chained Equations (MICE).\" \"The function generates an m number of imputed datasets [m = 5 in this study (43)] which differ in the imputed values. Then, a binary logistic regression was conducted using the data of all the imputed datasets.\""
  },
  "model": {
   "family": "logistic",
   "weighted": null,
   "quote": "Statistical analysis: \"A logistic regression model was employed to assess the association between six systemic immune biomarkers and MASLD. Three models were analyzed to enhance the robustness of the results. Model 1 was the unadjusted model.\" Weighting is never mentioned (see design.weights)."
  },
  "unstated": [
   "Whether survey weights, strata and PSUs were used (never mentioned; the SPSS output suggests unweighted)",
   "Which triglyceride measurement (fasting-subsample TRIGLY vs non-fasting BIOPRO) entered FLI, and whether the sample was restricted to the fasting subsample (USFLI needs fasting insulin and glucose); how participants with FLI but no USFLI (or the reverse) were classified",
   "The USFLI insulin term: printed as '1.1792∗insulin' (the published index uses ln(insulin)); units of insulin and glucose",
   "Units of FLI inputs (TG mg/dL, GGT U/L, waist cm are the published ones)",
   "How the MASLD exclusions were applied (dropped vs counted as non-MASLD) and their counts; which drugs; how autoimmune liver disease was identified before 2017-2018 (the liver-condition type items MCQ510a-f are in the 2017-2018 file but not in the 2011-2012 or 2015-2016 files); how the alcohol limits were computed",
   "What 'Missing data of liver (n = 20,000)' means (which variables), given the round number",
   "Thresholds for the binary TG and HDL covariates; physical activity domains; reference categories of race/ethnicity and marital status",
   "How the five imputed datasets were pooled, and the model selection mentioned in 'we chose the model that best adjusted to the data'",
   "Whether quartile cutpoints are sample quartiles of the 14,413 (cutpoints are printed in the Figure 3 legend)"
  ],
  "notes": "Fatty liver is defined by serum/anthropometric indices (FLI >= 60 or USFLI >= 30), not elastography, so 2021-2023 can build it from TRIGLY_L/BIOPRO_L, BMX_L, INS_L and GLU_L (fasting subsample); LUX_L is not needed. E5 judgment call: the MASLD definition's exclusion of secondary liver disease from drugs is treated as a medication exclusion step (brief: exclusion steps do not decide eligibility); if the study lead counts it as a component of the disease definition instead, E5 would fail on it, since 2021-2023 has no drug names (the viral, alcohol, liver cancer and autoimmune hepatitis exclusions are all constructible). Model 3 adjusts for BMI, WC, TG and GGT, which are inputs of the outcome indices (Table S4(A): BMI >= 30 OR 229.7, WC OR 5.8, TG OR 9.8), so the headline is conditional on most of the outcome's own components. The text never mentions survey weights; the SPSS outputs look unweighted. Internal inconsistencies: (1) Results say 'Age < 20 (n = 2,572)' while the flowchart says 25072 (= 59,842 - 34,770); 34,770 - 20,000 - 57 = 14,713, not 14,413. (2) Non-MASLD is 7,985 in the text and flowchart but 7,895 in Table 1, and the Results also say '6,518 had non-MASLD, 7,895 had MASLD' (reversed). (3) Table 1 shows diabetes 'Yes' 80.53%, cardiovascular disease 'Yes' 74.25%, hypertension 'Yes' higher in non-MASLD, and PIR < 1.3 at 67.38%, which look like reversed labels; PLR means of 0.01 imply a unit error; NPAR (neutrophil % / albumin) values around 1.37 imply albumin in g/L although the text says g/dL. (4) The abstract's Q4 vs Q1 ORs for NLR (1.25, 1.04-1.49) and PLR (1.29, 1.09-1.53) do not match Figure 3 Model 3 (NLR 1.29 (1.09 to 1.52); PLR 1.13 (0.96 to 1.33), not significant), and SIRI/NPAR differ slightly; the SII value matches Figure 3 and Table S4(A) exactly. (5) The propensity-matching paragraph's ORs (SII 1.62 (1.48, 1.78), SIRI 1.92, NLR 1.48, NPAR 2.04) equal Figure 3's crude full-sample Model 1 values, not the matched-sample Table S5 results (SII(3) 1.556 (1.277, 1.897)). (6) Results call the Q4 vs Q1 ORs 'per standard deviation'. (7) Figure 4 and 5 legends mention 'CDAI, composite dietary antioxidant index' (copy-paste remnant). SPSS indicator coding SII(1)-(3) = Q2-Q4 vs Q1 (consistent with Figure 3). Supplement (Table_1.DOCX, Tables S1-S6) fetched from the Europe PMC supplementaryFiles endpoint and Figures 1 and 3 from the PMC image CDN into scratchpad/dl/supp/row327/; the Figure 1 and Figure 3 numbers in this file are transcriptions of those images. Not a 2017-March 2020 paper.",
  "adjudication": null
 },
 {
  "id": "row034",
  "rank": 99,
  "row": 34,
  "doi": "10.1186/s12889-023-17556-w",
  "pmcid": "PMC10763382",
  "title": "Association between dietary inflammatory index and Stroke in the US population: evidence from NHANES 1999–2018",
  "authors": [
   "Mao, Yukang",
   "Weng, Jiayi",
   "Xie, Qiyang",
   "Wu, Lida",
   "Xuan, Yanling",
   "Zhang, Jun",
   "Han, Jun"
  ],
  "year": 2024,
  "journal": "BMC Public Health",
  "table_a": {
   "predictor": "Dietary inflammatory index",
   "condition": "Stroke",
   "population": "US population"
  },
  "headline": {
   "abstract_quote": "After confounder adjustment, the adjusted odds ratios (ORs) with 95% confidence intervals (CIs) for stroke across higher DII quartiles were 1.19 (0.94–1.54), 1.46 (1.16–1.84), and 1.87 (1.53–2.29) compared to the lowest quartile, respectively.",
   "table_location": "Table 3, Model II, row Q4",
   "table_quote": "| Q4 | 2.47 [2.08, 2.94] | < 0.001*** | 2.43 [2.02, 2.93] | < 0.001*** | 1.87 [1.53, 2.29] | < 0.001*** |",
   "measure": "OR",
   "estimate": 1.87,
   "ci_low": 1.53,
   "ci_high": 2.29,
   "p_value": "< 0.001***",
   "exposure_contrast": "DII quartile 4 vs quartile 1 (reference)",
   "model_label": "Model II",
   "covariates_in_this_model": [
    "age",
    "sex",
    "race/ethnicity",
    "educational level",
    "smoking status",
    "alcohol consumption",
    "BMI",
    "diabetes",
    "hypertension"
   ],
   "n_analytic": 44019,
   "n_quote": "Abstract: \"We collected the cross-sectional data of 44,019 participants of the National Health and Nutrition Examination Survey (NHANES) 1999–2018.\" Study population: \"After manual data filtration, we ultimately selected a total of 44,019 participants for subsequent analyses.\" (No separate N for Model II; handling of missing covariates is not described.)",
   "events": 1486
  },
  "cycles": [
   "1999-2000",
   "2001-2002",
   "2003-2004",
   "2005-2006",
   "2007-2008",
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": "18-79 (excluded < 18 or >= 80); effectively 20-79 because stroke status is asked of ages 20+",
   "defining": "US general population of adults, not pregnant",
   "inclusion": "NHANES 1999-2018 participants aged 18-79, not pregnant, with dietary intake and stroke data",
   "exclusions": [
    "Aged < 18 or >= 80 years (n = 46,469)",
    "Pregnant (n = 1,516)",
    "Without relevant information on dietary intake (n = 5,747)",
    "Without stroke status (n = 3,665)"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "Study population: \"The exclusion criteria were set as follows: (1) participants aged < 18 or ≥ 80 years (n = 46,469); (2) participants who were pregnant (n = 1,516); (3) participants without relevant information on dietary intake (n = 5,747) and stroke status (n = 3,665). After manual data filtration, we ultimately selected a total of 44,019 participants for subsequent analyses.\""
  },
  "exposure": {
   "definition": "Dietary Inflammatory Index (Shivappa et al. 2014) from the 24-hour dietary recall taken in the MEC (day 1), using 26 of the 45 food parameters; the component scores are summed (positive = pro-inflammatory). Analyzed in quartiles at fixed cutpoints and per 1 unit (continuous).",
   "nhanes_variables": [
    "DR1TKCAL",
    "DR1TPROT",
    "DR1TCARB",
    "DR1TFIBE",
    "DR1TTFAT",
    "DR1TSFAT",
    "DR1TMFAT",
    "DR1TPFAT",
    "DR1TCHOL",
    "DR1TVARA",
    "DR1TBCAR",
    "DR1TVB1",
    "DR1TVB2",
    "DR1TNIAC",
    "DR1TVB6",
    "DR1TFOLA",
    "DR1TVB12",
    "DR1TVC",
    "DR1TVD",
    "DR1TATOC",
    "DR1TMAGN",
    "DR1TIRON",
    "DR1TZINC",
    "DR1TSELE",
    "DR1TCAFF",
    "DR1TALCO"
   ],
   "nhanes_files": [
    "DR1TOT",
    "DRXTOT"
   ],
   "transform": "categories: quartiles at fixed cutpoints (Q1 < 0.23; Q2 0.23 to < 1.76; Q3 1.76 to < 2.95; Q4 >= 2.95), Q1 reference",
   "categories": "Statistical analysis: \"The DII scores were categorized into four quartiles (Q1: DII < 0.23; Q2: 0.23 ≤ DII < 1.76; Q3: 1.76 ≤ DII < 2.95; Q4: DII ≥ 2.95), with the first quartile (Q1) being designated as the reference quartile.\"",
   "quote": "Assessment of dietary information: \"Dietary intake data regarding the types and amounts of food and drinks consumed during the 24-hour period prior to the interview was recorded in the mobile examination center (MEC), and was used to calculate DII as we previously reported [29, 30].\" \"In the present study, 26 of 45 food parameters were incorporated: energy, protein, carbohydrate, dietary fiber, total fat, saturated fat, monounsaturated fatty acids (MUFAs), polyunsaturated fatty acids (PUFAs), cholesterol, vitamin A, β carotene, vitamin B1, vitamin B2, niacin, vitamin B6, total folate, vitamin B12, vitamin C, vitamin D, vitamin E, magnesium, iron, zinc, selenium, caffeine, and alcohol. To evaluate the inflammatory potentials of one participant’s diet, all food component-specific DII scores were then summed to yield an overall DII\""
  },
  "outcome": {
   "definition": "Self-reported stroke: yes to ever told by a physician or health professional of a stroke",
   "nhanes_variables": [
    "MCQ160F"
   ],
   "nhanes_files": [
    "MCQ"
   ],
   "quote": "Assessment of Stroke: \"Stroke was defined by self-reported previous diagnosis by a physician during face-to-face interview. Anyone who answered “yes” to the following question: “Have you ever been told by a physician or a health professional that you had stroke?” was considered having stroke.\""
  },
  "covariates": [
   {
    "name": "age",
    "coding": "years (coding in Model II not stated; Table 1 reports a mean)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "sex",
    "coding": "male/female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "race/ethnicity",
    "coding": "5 categories: non-Hispanic White, non-Hispanic Black, other Hispanic, Mexican American, other races",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Reference category not stated."
   },
   {
    "name": "educational level",
    "coding": "below high school, high school, above high school",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DMDEDUC2 covers 20+ (18-19-year-olds had DMDEDUC3 in earlier cycles); the stroke item limits the analysis to 20+ in practice."
   },
   {
    "name": "smoking status",
    "coding": "smoker = smoked over 100 cigarettes in lifetime (current or former), else non-smoker",
    "nhanes_variables": [
     "SMQ020"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "alcohol consumption",
    "coding": "drinker = at least 12 drinks during the year preceding the survey, else non-drinker",
    "nhanes_variables": [
     "ALQ101",
     "ALQ120Q",
     "ALQ121",
     "ALQ130"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "Item not named. If 'the year preceding the survey' means the past 12 months, 2021-2023 ALQ121 (frequency) and ALQ130 (drinks per drinking day) can count it; the 1999-2016 item ALQ101 ('at least 12 alcohol drinks in any one year') is not in 2017-2018 or 2021-2023."
   },
   {
    "name": "BMI",
    "coding": "kg/m2; coding in Model II not stated (subgroups use normal weight, overweight, obesity with > 25 and > 30 as cut-offs)",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "diabetes",
    "coding": "diagnosed diabetes (self-report) or undiagnosed: HbA1c >= 6.5%, fasting plasma glucose >= 126 mg/dL, or 2-hour OGTT glucose >= 200 mg/dL",
    "nhanes_variables": [
     "DIQ010",
     "LBXGH",
     "LBXGLU",
     "LBXGLT"
    ],
    "nhanes_files": [
     "DIQ",
     "GHB",
     "GLU",
     "OGTT"
    ],
    "in_2021_2023": true,
    "note": "No OGTT in 2021-2023 (that component cannot be built); fasting glucose is in the fasting subsample only."
   },
   {
    "name": "hypertension",
    "coding": "any of: average SBP >= 140, average DBP >= 90 (mean of three readings), self-reported diagnosis, current anti-hypertensive drugs",
    "nhanes_variables": [
     "BPXSY1",
     "BPXSY2",
     "BPXSY3",
     "BPXDI1",
     "BPXDI2",
     "BPXDI3",
     "BPQ020",
     "BPQ050A"
    ],
    "nhanes_files": [
     "BPX",
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "2021-2023: BPXO_L (oscillometric) and BPQ020, BPQ150."
   }
  ],
  "design": {
   "weights": "Survey-weighted; the weight is not named: 'the sample weights corresponding to different research periods'. With day-1 dietary data the usual choice is WTDRD1 (with the 4-year weight for 1999-2002, combined across 10 cycles), but the paper does not say.",
   "strata_psu": "not stated",
   "quote": "Statistical analysis: \"Since NHANES survey employed a series of complex sampling designs, we took into account the sample weights corresponding to different research periods in our analytic methods to yield accurate estimates of health-related statistics [40–42].\"",
   "software": "R 'version 4.1.6' as printed (no such R release exists; 4.1.x ended at 4.1.3)",
   "missing_data": "unstated (participants without dietary or stroke data were excluded; missing covariates are not discussed)",
   "quote_missing": "Study population: \"(3) participants without relevant information on dietary intake (n = 5,747) and stroke status (n = 3,665).\" Nothing on missing covariates."
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "Abstract (Methods): \"The association of DII with stroke was estimated using weighted multivariate logistic regression, with its nonlinearity being examined by restricted cubic spline (RCS) regression.\" Results: \"As shown in Table 3, we performed a sampling-weighted multivariate logistic regression analysis for detecting the association between DII and stroke\"; Table 3 title: \"Table 3 Weighted logistic regression analysis on the association between DII and stroke\""
  },
  "unstated": [
   "Which weight (dietary day-1 WTDRD1 or MEC), how the 1999-2002 4-year weights and ten cycles were combined, and whether strata and PSUs were used",
   "Whether only the day-1 recall was used (implied by 'recorded in the mobile examination center') and whether supplement intake was excluded from the DII",
   "How the DII was computed in cycles lacking components: NHANES dietary files have vitamin D only from 2007-2008 (DR1TVD), and 1999-2000 has carotene and vitamin A in RE and vitamin E as ATE instead of beta-carotene, RAE and alpha-tocopherol",
   "Which global means/SDs and inflammatory weights were used (only 'as we previously reported')",
   "Coding of age and BMI in Model II (continuous or categories), and reference categories",
   "Which alcohol item defines 'at least 12 drinks during the year preceding the survey' in each cycle",
   "Handling of missing covariates and the Model II analytic N",
   "Diabetes: OGTT and fasting glucose exist only in subsamples/some cycles; how missing lab values were treated",
   "Hypertension: averaging when fewer than three readings, use of the fourth reading in 1999-2018",
   "How pregnancy was determined (RIDEXPRG or urine test)",
   "Whether the quartile cutpoints were weighted or unweighted sample quartiles"
  ],
  "notes": "Headline is Q4 vs Q1 (Model II, weighted) under the clarified rule for ordered categories (highest vs lowest reference); Q2 vs Q1 is 1.19 (0.94-1.54, P 0.15), Q3 vs Q1 1.46 (1.16-1.84), and the continuous DII 1.15 (1.10, 1.20). The unweighted sensitivity analysis (Supplementary Table 3) gives Q4 vs Q1 1.89 (1.60, 2.23) for Model II, so the main Table 3 is the weighted analysis. Table 3's Model I Q2 shows CI 1.01-1.58 with P 0.002**, inconsistent with the CI. Table 1 is garbled: the rows WBC, NE, LY, PLT, Hemoglobin repeat the values of Age, Sex, Non-Hispanic White, Non-Hispanic Black and Mexican American; TC repeats Monocyte; units and values for FBG, HbA1c, eGFR, TG, HDL-C, RBC are implausible. Table 2 vitamin E means (66.6 mg) are implausible for alpha-tocopherol. The LASSO/nomogram analysis is not relevant. Weighted analysis is stated, weight not named. 2021-2023 has all 26 DII components (DR1TOT_L) but both recalls were by telephone. The diabetes covariate's OGTT component and the 1999-2016 alcohol item ALQ101 are not in 2021-2023. Supplementary Tables 1-3 (docx) were fetched from Europe PMC to scratchpad/dl/supp/row034/; Supplementary Table 1 (baseline by DII quartile) has plausible values for the rows garbled in the main Table 1. Not a 2017-March 2020 paper.",
  "adjudication": "Under the headline rule as settled, a coding with ordered categories and the lowest as reference is represented by its highest category against the reference; the extraction took the first significant contrast the abstract listed."
 },
 {
  "id": "row028",
  "rank": 104,
  "row": 28,
  "doi": "10.31083/J.RCM2504130",
  "pmcid": "PMC11264029",
  "title": "Systemic Immune-Inflammation Index and Its Association with the Prevalence of Stroke in the United States Population: A Cross-Sectional Study Using the NHANES Database",
  "authors": [
   "Liu, Guangcheng",
   "Qian, Hao",
   "Wang, Liang",
   "Wu, Wei"
  ],
  "year": 2024,
  "journal": "Reviews in Cardiovascular Medicine",
  "table_a": {
   "predictor": "Systemic immune-inflammation index",
   "condition": "Stroke",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "The high SII group had a substantially greater prevalence of stroke compared to the low SII group (odds ratio [OR] = 1.18, 95% confidence interval [CI] 1.01, 1.42).",
   "table_location": "Table 2 (Logistic regression analysis for the risk of stroke according to SII in the overall NHANES participant group), row 'Model 3', column 'OR (95% CI) High' (reference 'Low' = 1.00)",
   "table_quote": "| Model | Per one unit increase in log-transformed SII OR (95% CI) | OR (95% CI) <colspan=4> |\n| Low | Median | High | p trend | [...] | Model 3 | 1.30 (0.99, 1.70) | 1.00 | 1.10 (0.92, 1.31) | 1.18 (1.01, 1.42) | 0.041 |",
   "measure": "OR",
   "estimate": 1.18,
   "ci_low": 1.01,
   "ci_high": 1.42,
   "p_value": "not printed for the contrast (Table 2, Model 3 p trend 0.041)",
   "exposure_contrast": "SII-high tertile (SII >= 597) vs SII-low tertile (SII < 384, reference); SII = platelet count x neutrophil count / lymphocyte count, counts in 10^3 cells/uL",
   "model_label": "Model 3 (fully adjusted model)",
   "covariates_in_this_model": [
    "age",
    "sex",
    "race/ethnicity",
    "smoking status",
    "physical activity",
    "education level",
    "family income to poverty ratio",
    "BMI",
    "diabetes",
    "dyslipidemia",
    "cancer",
    "hypertension"
   ],
   "n_analytic": 57600,
   "n_quote": "The final analysis included 57,600 participants, of which 2368 were stroke patients",
   "events": 2368
  },
  "cycles": [
   "1999-2020 (span only; the paper does not list the cycles)"
  ],
  "population": {
   "age": ">=18",
   "defining": "US adults aged 18 or above (both sexes; no condition named)",
   "inclusion": "NHANES 1999-2020 adult participants aged 18 or above (66,568) with SII, stroke, smoking status, hyperlipidemia and hypertension data.",
   "exclusions": [
    "lacked information on SII data (n = 6836)",
    "missing data on stroke diagnosis (n = 2059)",
    "smoking status (n = 47)",
    "hyperlipidemia (n = 2)",
    "hypertension (n = 24)"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "This study included adults aged 18 or above in the NHANES database and spanning the years from 1999 to 2020. Initially, a total of 66,568 adult participants were enrolled. Participants were excluded if they lacked information on SII data (n = 6836) or had missing data on stroke diagnosis (n = 2059), smoking status (n = 47), hyperlipidemia (n = 2), and hypertension (n = 24). The final analysis included 57,600 participants, of which 2368 were stroke patients (refer to Fig. 1 for the participant flowchart). The participants were categorized into three groups based on SII tertiles: SII-low (<384), SII-median (384–597), and SII-high (≥597)."
  },
  "exposure": {
   "definition": "Systemic immune-inflammation index: platelet count x neutrophil count / lymphocyte count, counts from the automated hematology analyzer in 10^3 cells/uL (the paper prints '× 103 cells/mL'). Tertiles for the headline; log-transformed for the continuous analysis (log base not stated).",
   "nhanes_variables": [
    "LBXPLTSI",
    "LBDNENO",
    "LBDLYMNO"
   ],
   "nhanes_files": [
    "CBC"
   ],
   "transform": "tertiles (headline: highest vs lowest); log-transformed SII per 1 unit for the continuous model",
   "categories": "SII-low (<384), SII-median (384-597), SII-high (>=597). Quote (2.1): 'The participants were categorized into three groups based on SII tertiles: SII-low (<384), SII-median (384–597), and SII-high (≥597).'",
   "quote": "SII was calculated using the formula: (platelet count × neutrophil count)/lymphocyte count. Lymphocyte, neutrophil, and platelet counts were measured using automated hematology analyzers and expressed as × 103 cells/mL [11]."
  },
  "outcome": {
   "definition": "Self-reported stroke ever diagnosed by a doctor or other health professional (yes/no).",
   "nhanes_variables": [
    "MCQ160F"
   ],
   "nhanes_files": [
    "MCQ"
   ],
   "quote": "The primary outcome measure in this study was self-reported stroke, as determined by asking participants: “Have you ever been told you had a stroke?”, or “Has a doctor or other health professional ever told you that you had a stroke?”."
  },
  "covariates": [
   {
    "name": "age",
    "coding": "continuous",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L RIDAGEYR."
   },
   {
    "name": "sex",
    "coding": "male or female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L RIAGENDR."
   },
   {
    "name": "race/ethnicity",
    "coding": "non-Hispanic White, non-Hispanic Black, Mexican American, and others",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L RIDRETH1 ('others' = Other Hispanic plus Other race, presumably)."
   },
   {
    "name": "smoking status",
    "coding": "never (<100 cigarettes in life), former (>100 cigarettes, quit), current (>100 cigarettes and smokes some days or every day) (Supplementary Method 1)",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "SMQ_L SMQ020, SMQ040."
   },
   {
    "name": "physical activity",
    "coding": "sedentary (MET-minutes/week = 0), insufficient (0-500), moderate (500-1000), high (>1000), from weekly minutes of moderate and vigorous activities multiplied by MET (Supplementary Method 1)",
    "nhanes_variables": [
     "PAD200",
     "PAD320",
     "PAQ605",
     "PAQ620",
     "PAQ635",
     "PAQ650",
     "PAQ665"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": false,
    "note": "Domains not stated. The 1999-2006 PAQ (PAD200/PAD320 leisure moderate/vigorous plus other items) and the 2007-2018 GPAQ (work, transport, recreation) differ; PAQ_L has only leisure-time moderate/vigorous frequency and minutes (PAD790Q/U, PAD800, PAD810Q/U, PAD820), so a leisure-only MET-minutes version can be built but not one including work or transport."
   },
   {
    "name": "education level",
    "coding": "under high school, high school or equivalent, college or higher",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L DMDEDUC2 (ages 20+; no DMDEDUC3 for 18-19-year-olds in 2021-2023)."
   },
   {
    "name": "family income to poverty ratio",
    "coding": "<=1.0, 1.0-3.0, >3.0, unknown (missing as its own category)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L INDFMPIR."
   },
   {
    "name": "BMI",
    "coding": "<25.0, 25.0-29.9, >=30.0 kg/m2",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "BMX_L BMXBMI; missing BMI (1.72%) median-imputed."
   },
   {
    "name": "diabetes",
    "coding": "yes/no: self-reported medical history, use of oral hypoglycemic agents or insulin, fasting glucose >=126 mg/dL, or HbA1c >=6.5%",
    "nhanes_variables": [
     "DIQ010",
     "DIQ070",
     "DIQ050",
     "LBXGLU",
     "LBXGH"
    ],
    "nhanes_files": [
     "DIQ",
     "GLU",
     "GHB"
    ],
    "in_2021_2023": true,
    "note": "DIQ_L DIQ010, DIQ070, DIQ050; GLU_L LBXGLU (fasting subsample); GHB_L LBXGH."
   },
   {
    "name": "dyslipidemia (hyperlipidemia)",
    "coding": "yes/no: triglycerides >=150 mg/dL, total cholesterol >=200 mg/dL, LDL-C >=130 mg/dL, HDL-C <=40 mg/dL in men or <=50 mg/dL in women, or use of medication for hyperlipidemia",
    "nhanes_variables": [
     "LBXTR",
     "LBXTC",
     "LBDLDL",
     "LBDHDD",
     "BPQ100D",
     "BPQ101D"
    ],
    "nhanes_files": [
     "TRIGLY",
     "TCHOL",
     "HDL",
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "TRIGLY_L LBXTLG and LBDLDL (fasting subsample), TCHOL_L LBXTC, HDL_L LBDHDD, BPQ_L BPQ101D (taking cholesterol medication)."
   },
   {
    "name": "cancer",
    "coding": "yes/no: 'Have you ever been told you had cancer or malignancy?'",
    "nhanes_variables": [
     "MCQ220"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "MCQ_L MCQ220. Listed in the Table 2 note for Model 3 but not in the Methods' Model 3 description."
   },
   {
    "name": "hypertension",
    "coding": "yes/no; definition not given",
    "nhanes_variables": [
     "BPQ020"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Definition not stated. Self-report (BPQ_L BPQ020), medication (BPQ150) and oscillometric BP (BPXO_L) exist in 2021-2023; earlier cycles measured BP by auscultation (BPX), so a measured-BP definition would change device."
   }
  ],
  "design": {
   "weights": "'appropriate weights' (which weight, and how 2-year weights were combined across cycles, not stated)",
   "strata_psu": "not stated by name; R 'survey' package used",
   "quote": "In this study, appropriate weights were employed to account for the complex sampling design of NHANES, thereby ensuring a representative sample of the US national population (https://www.cdc.gov/nchs/nhanes/index.htm). [...] All statistical analyses were conducted using R version 4.1.3 (R Foundation for Statistical Computing, Vienna, Austria) with the “survey” package.",
   "software": "R version 4.1.3 with the 'survey' package",
   "missing_data": "Exclusion for missing SII, stroke, smoking status, hyperlipidemia or hypertension; missing income-to-poverty ratio as an 'Unknown' category; median imputation for other covariates (e.g., BMI, 1.72% missing)",
   "quote_missing": "The percentage of missing data for covariates was <5% (body mass index (BMI) [1.72%]), except for the family income to poverty ratio (9.4%). Missing values for the family income to poverty ratio were categorized as “Unknown”. To incorporate all available data for modeling, imputation with the median of each variable was performed."
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "For the analysis, odds ratios (ORs) and 95% confidence intervals (CIs) were estimated using multivariate logistic regression models to assess the association between SII and the prevalence of stroke. When treating SII as a continuous variable, a log-transformation was applied and the change in stroke prevalence for each one-unit increase in log-transformed SII was determined. Restricted cubic spline (RCS) analysis in the fully adjusted model was employed to evaluate the dose-response relationship between SII and stroke, with nonlinearity assessed using the likelihood ratio test."
  },
  "unstated": [
   "which cycles were pooled: only the span 1999-2020 is given; '2020' implies the 2017-March 2020 pre-pandemic files, but whether those were used, and if so whether the 2017-2018 files were also included (they overlap), is not stated",
   "which weight (MEC 2-year, 4-year for 1999-2002, pre-pandemic) and how weights were combined across cycles",
   "strata and PSU variable names",
   "the log base of log-transformed SII",
   "whether the SII tertile cutpoints were computed weighted or unweighted",
   "the hypertension definition",
   "physical activity: questionnaire items, domains and MET values across the 1999-2006 PAQ and 2007-2018 GPAQ",
   "how diabetes and lipid medications were identified (questionnaire items or prescription file)",
   "handling of fasting glucose, triglycerides and LDL-C (fasting subsample only) for non-fasting participants",
   "whether cancer was in Model 3 (Table 2 note: yes; Methods: not listed)",
   "how 18-19-year-olds were handled: the stroke question is asked from age 20, so they lack the outcome",
   "reference categories of covariates"
  ],
  "notes": "Sample size is printed three ways: Abstract '53,600 people', Introduction and Methods 57,600, Discussion '53,111 participants'; Table 1 and Table 2 use 57,600 with 2368 strokes. The abstract's continuous estimate is misworded: 'The risk of stroke decreased by 34% for every unit rise in log-transformed SII (OR 1.30, 95% CI 0.99, 1.70)' gives Model 3's non-significant OR with Model 2's 34% (OR 1.34) and the wrong direction. The headline CI is asymmetric on the log scale (ln 1.18 = 0.166; ln 1.01 = 0.010; ln 1.42 = 0.351), which may reflect a typo in one bound; the abstract and Table 2 agree on 1.18 (1.01, 1.42). Model 3 covariates differ between the Methods ('diabetes, hyperlipidemia, and hypertension') and the Table 2 note ('diabetes, dyslipidemia, cancer, and hypertension'); covariates_in_this_model follows the table note. The Methods print counts as '× 103 cells/mL' (10^3 cells/uL in NHANES). Table 1's race rows and Table 3's cancer 'No' row (4871/52,200, more strokes than the 2368 total) are internally inconsistent. Fig. 2's legend mentions 'HR'. Because MCQ160f is asked only from age 20, the 2059 excluded for missing stroke presumably include the 18-19-year-olds (our inference). Supplementary Method 1 (covariate coding) was fetched from Europe PMC to scratchpad/dl/supp/row028/ (zip with the docx and the two figures); Figure 1 there says 'Adult participants in the NHANES 1999-2020 (N=66568)' and lists the same exclusions. NHANES variable names are our mapping; the paper names none.",
  "adjudication": null
 },
 {
  "id": "row319",
  "rank": 105,
  "row": 319,
  "doi": "10.3390/antiox12091740",
  "pmcid": "PMC10525155",
  "title": "Association between the Composite Dietary Antioxidant Index and Atherosclerotic Cardiovascular Disease in Postmenopausal Women: A Cross-Sectional Study of NHANES Data, 2013–2018",
  "authors": [
   "Liu, Chenning",
   "Lai, Wenyu",
   "Zhao, Meiduo",
   "Zhang, Yexuan",
   "Hu, Yuanjia"
  ],
  "year": 2023,
  "journal": "Antioxidants",
  "table_a": {
   "predictor": "Composite dietary antioxidant index",
   "condition": "Cardiovascular disease",
   "population": "US adult postmenopausal females"
  },
  "headline": {
   "abstract_quote": "The ORs associated with a per-SD increase in CDAI were 0.67 (95% CI: 0.51–0.88) for ASCVD risk.",
   "table_location": "Table 2 (Associations between CDAI Levels and the Risks of ASCVD and Hard Criteria in Postmenopausal Women), row 'As continuous (per SD)', column 'Model C ASCVD'",
   "table_quote": "| CDAI | Model A <colspan=2> | Model B <colspan=2> | Model C <colspan=2> |\n| ASCVD | Hard Criteria b | ASCVD | Hard Criteria b | ASCVD | Hard Criteria b | [...] | As continuous (per SD) | 0.65 (0.53, 0.80) *** | 0.61 (0.50, 0.75) *** | 0.71 (0.58, 0.87) *** | 0.67 (0.54, 0.83) *** | 0.67 (0.51, 0.88) ** | 0.61 (0.46, 0.81) *** |",
   "measure": "OR",
   "estimate": 0.67,
   "ci_low": 0.51,
   "ci_high": 0.88,
   "p_value": "** (p ≤ 0.01; exact p not printed)",
   "exposure_contrast": "per 1 SD increase in CDAI (SD not reported; CDAI is a sum of six standardized nutrient intakes, median -0.18 [IQR -2.27, 2.46] in Table 1)",
   "model_label": "Model C",
   "covariates_in_this_model": [
    "age",
    "race",
    "education level",
    "marital status",
    "the ratio of family income to poverty",
    "BMI",
    "waist circumference",
    "alcohol use",
    "smoking—cigarette use",
    "moderate to vigorous recreational activities",
    "sleep disorders",
    "hypertension",
    "diabetes",
    "a family history of heart attack",
    "NLR",
    "HDL-C",
    "total cholesterol",
    "total daily caloric intake",
    "total daily polyunsaturated fatty acids"
   ],
   "n_analytic": 3109,
   "n_quote": "In total, 3109 menopausal women were included in our analysis, of whom 453 were diagnosed with ASCVD and whose age range was 40 to 80 years old, with an average age of 64.66 ± 9.06 years.",
   "events": 453
  },
  "cycles": [
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": "no age limit set; observed 40 to 80 (age top-coded at 80); women without menstrual information enter only from age 55",
   "defining": "postmenopausal women: RHD043 answer 'Menopause/Change of life', or women aged >=55 with no menstrual information",
   "inclusion": "NHANES 2013-2018 women classified as postmenopausal (4057), with dietary data, ASCVD information and income-to-poverty ratio.",
   "exclusions": [
    "missing dietary information (n = 544)",
    "missing ASCVD information (CHD, angina, heart attack, stroke) (n = 34)",
    "missing ratio of family income to poverty (n = 370)"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "In total, 29,400 individuals participated in the NHANES 2013–2018 cycle, for which menopausal status was determined based on self-reported menstrual information and age. The newest RHD043 item was added to the NHANES Reproductive Health Questionnaire in 2013, and it requests the reasons for the absence of menstrual periods over the past 12 months. Participants who responded “Menopause/Change of life” were classified as postmenopausal status, and women aged ≥55 years old with no menstrual information were also included in the postmenopausal group, leading to 4057 postmenopausal women in our study, in total. Participants with missing dietary (n = 544) or ASCVD [including coronary heart disease (CHD), angina, heart attack, and stroke, n = 34] information were excluded from the analysis, and covariates with missing values <10% were imputed using the random forest method. However, those with missing values ≥10%, such as the ratio of family income to poverty, were not imputed, and participants (n = 370) with missing information on this variable were further excluded. Therefore, our final analysis included 3109 postmenopausal women (Figure 1)."
  },
  "exposure": {
   "definition": "Composite dietary antioxidant index (modified Wright et al.): CDAI = sum over zinc, selenium, carotenoids, vitamin A, vitamin C and vitamin E of (individual daily average intake - mean)/SD, where daily average intake is the mean of the two 24-hour dietary recalls. Food intake only: supplements were not considered (Discussion: 'we did not consider the difference between daily dietary intakes of vitamins or minerals and additional supplementation (such as from medication) of vitamins or minerals.').",
   "nhanes_variables": [
    "DR1TZINC",
    "DR2TZINC",
    "DR1TSELE",
    "DR2TSELE",
    "DR1TACAR",
    "DR2TACAR",
    "DR1TBCAR",
    "DR2TBCAR",
    "DR1TCRYP",
    "DR2TCRYP",
    "DR1TLYCO",
    "DR2TLYCO",
    "DR1TLZ",
    "DR2TLZ",
    "DR1TVARA",
    "DR2TVARA",
    "DR1TVC",
    "DR2TVC",
    "DR1TATOC",
    "DR2TATOC"
   ],
   "nhanes_files": [
    "DR1TOT",
    "DR2TOT"
   ],
   "transform": "per SD (headline); also high vs low by an unstated cut-off, and quartiles",
   "categories": "Quartiles (Table 2 and Abstract): Q1 (−6.83–−1.04), Q2 (−1.04–1.11), Q3 (1.11–3.72), Q4 (3.72–43.87); the high/low cut-off is not stated.",
   "quote": "The NHANES collected participants’ food intake data through nonconsecutive two-day 24 h-dietary recall interviews. The first dietary recall interview was conducted in person at the Mobile Examination Center (MEC), while the second was conducted over the phone 3 to 10 days later. The daily average intakes were calculated from these two days’ dietary recall data, and we computed the CDAI levels for all the subjects using a modified version developed by Wright et al. [17]. The CDAI was the sum of the daily average intakes of zinc, selenium, carotenoids, vitamin A, vitamin C, and vitamin E, which were first normalized by subtracting the mean and then dividing by their standard deviation (SD):\n[FORMULA] CDAI=∑i=1n=6(IndividualIntake−Mean)/SD"
  },
  "outcome": {
   "definition": "ASCVD (binary): at least one self-reported diagnosis of coronary heart disease, angina, heart attack, or stroke.",
   "nhanes_variables": [
    "MCQ160C",
    "MCQ160D",
    "MCQ160E",
    "MCQ160F"
   ],
   "nhanes_files": [
    "MCQ"
   ],
   "quote": "The outcome of our research was ASCVD, defined as the presence of at least one diagnosis of CHD, angina, heart attack, or stroke according to the 2013 American College of Cardiology (ACC) and the American Heart Association (AHA) Guideline on the Treatment of Blood Cholesterol to Reduce Atherosclerotic Cardiovascular Risk in Adults [16]. Hard criteria were defined as a history of heart attack or stroke."
  },
  "covariates": [
   {
    "name": "age",
    "coding": "Table S1: 'For individuals who were 80 years and older, their age was topcoded at 80 years. Age was categorized into five groups: 40-49, 50-59, 60-69, 70-79, and ≥80 years.' (whether the model used the groups or continuous age is not stated)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L RIDAGEYR (top-coded at 80)."
   },
   {
    "name": "race",
    "coding": "Mexican American, other Hispanic, non-Hispanic white, non-Hispanic black, or other race",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L RIDRETH1."
   },
   {
    "name": "education level",
    "coding": "less than high school, high school or equivalent, or college or above",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L DMDEDUC2 (ages 20+)."
   },
   {
    "name": "marital status",
    "coding": "Table S1: 'Marital status was reported as married, widowed, divorced, separated, never married, or living with a partner.'",
    "nhanes_variables": [
     "DMDMARTL"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": false,
    "note": "DEMO_L releases only DMDMARTZ with 3 groups (1 Married/Living with partner, 2 Widowed/Divorced/Separated, 3 Never married), so the 6 groups cannot be rebuilt."
   },
   {
    "name": "the ratio of family income to poverty",
    "coding": "Table S1: 'Ratio of family income to poverty was divided into three groups: ≤1.00, 1.01-3.00, and >3.00.'",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L INDFMPIR; participants missing it were excluded (n = 370)."
   },
   {
    "name": "BMI",
    "coding": "Table S1: 'The body mass index (BMI) was calculated by dividing weight in kilograms by height in meters squared. It was categorized into four groups: underweight (<18.5 kg/m2), normal weight (18.5-24.9 kg/m2), overweight (25.0-29.9 kg/m2), and obesity (≥30 kg/m2).'",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "BMX_L BMXBMI."
   },
   {
    "name": "waist circumference",
    "coding": "cm (Table 1 median [IQR]); model coding not stated",
    "nhanes_variables": [
     "BMXWAIST"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "BMX_L BMXWAIST."
   },
   {
    "name": "alcohol use",
    "coding": "Table S1: 'Alcohol use was defined as consuming at least 12 drinks of any type of alcoholic beverage in any one year. Alcoholic beverages included liquor, beer, wine, wine coolers, and any other type of alcoholic beverage.'",
    "nhanes_variables": [
     "ALQ101"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": false,
    "note": "The 'at least 12 drinks in any one year' item (ALQ101) was asked in 2013-2016 but not in 2017-2018 (ALQ_J starts with ALQ111, ever had a drink) or 2021-2023 (ALQ_L: ALQ111, ALQ121, ...); how the paper coded 2017-2018 is not stated."
   },
   {
    "name": "smoking—cigarette use",
    "coding": "Table S1: 'Smoking behaviors were based on whether participants had smoked at least 100 cigarettes in their lifetime.' (yes/no)",
    "nhanes_variables": [
     "SMQ020"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "SMQ_L SMQ020."
   },
   {
    "name": "moderate to vigorous recreational activities",
    "coding": "Table S1: 'The Physical Activity questionnaire recorded whether participants engaged in moderate or vigorous recreational activities. Responses were categorized as “yes” or “no”.'",
    "nhanes_variables": [
     "PAQ650",
     "PAQ665"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": true,
    "note": "2013-2018 asked whether, in a typical week, the participant does vigorous (PAQ650) or moderate (PAQ665) recreational activity for at least 10 minutes; PAQ_L asks how often (PAD790Q/U moderate, PAD810Q/U vigorous leisure-time activity), so 'yes' = any moderate or vigorous LTPA, with changed wording."
   },
   {
    "name": "sleep disorders",
    "coding": "Table S1: 'Participants were asked whether they had ever told a doctor that they had trouble sleeping, and those who answered “yes” were classified as having sleep disorders.'",
    "nhanes_variables": [
     "SLQ050"
    ],
    "nhanes_files": [
     "SLQ"
    ],
    "in_2021_2023": false,
    "note": "SLQ_L has sleep times and hours only; no trouble-sleeping question."
   },
   {
    "name": "hypertension",
    "coding": "Table S1: 'Hypertension was defined based on self-reported information, either from a doctor’s diagnosis or advice to take antihypertensive medication.'",
    "nhanes_variables": [
     "BPQ020",
     "BPQ040A"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "BPQ040A ('told to take prescribed medicine' for high blood pressure) is asked only after a yes to BPQ020 (BPQ_J: BPQ020 No skips to BPQ080), so the definition reduces to BPQ020 = yes; BPQ_L BPQ020."
   },
   {
    "name": "diabetes",
    "coding": "Table S1: 'Diabetes was defined as self-reported diabetes (participants who answered “yes” to the question “Has a doctor told you that you have diabetes?”) or current use of hypoglycemic agents or insulin, or a hemoglobin A1c (HbA1c) level ≥ 6.5%.'",
    "nhanes_variables": [
     "DIQ010",
     "DIQ070",
     "DIQ050",
     "LBXGH"
    ],
    "nhanes_files": [
     "DIQ",
     "GHB"
    ],
    "in_2021_2023": true,
    "note": "DIQ_L DIQ010, DIQ070, DIQ050; GHB_L LBXGH."
   },
   {
    "name": "a family history of heart attack",
    "coding": "Table S1: 'A family history of heart attack was defined as a self-reported “yes” response to the question “Have any of your close biological relatives, including father, mother, sisters or brothers, been told by a health professional that they had a heart attack or angina before the age of 50?”.'",
    "nhanes_variables": [
     "MCQ300A"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": false,
    "note": "MCQ_L has no family-history items (no MCQ300A)."
   },
   {
    "name": "NLR",
    "coding": "Table S1: 'automated hematology analysis devices and expressed as ×1,000 cells/mm3. The neutrophil-to-lymphocyte ratio (NLR) was calculated as the ratio of the neutrophil count to the lymphocyte count.' (model coding not stated)",
    "nhanes_variables": [
     "LBDNENO",
     "LBDLYMNO"
    ],
    "nhanes_files": [
     "CBC"
    ],
    "in_2021_2023": true,
    "note": "CBC_L LBDNENO / LBDLYMNO."
   },
   {
    "name": "HDL-C",
    "coding": "not stated; Table 1 prints 'HDL-C (mg/dL) | 1.00 [0.00, 1.00]', which looks like a binary indicator (the stratified analysis splits at 50 mg/dL)",
    "nhanes_variables": [
     "LBDHDD"
    ],
    "nhanes_files": [
     "HDL"
    ],
    "in_2021_2023": true,
    "note": "HDL_L LBDHDD."
   },
   {
    "name": "total cholesterol",
    "coding": "mg/dL (Table 1 median [IQR]); model coding not stated",
    "nhanes_variables": [
     "LBXTC"
    ],
    "nhanes_files": [
     "TCHOL"
    ],
    "in_2021_2023": true,
    "note": "TCHOL_L LBXTC."
   },
   {
    "name": "total daily caloric intake",
    "coding": "Table 1 categories <1550, 1550–1972, 1973–2554, ≥2555 kcal/day; Table S1: 'Total daily caloric intake was estimated by analyzing the types and amounts of food and beverages (including all types of water) consumed during the 24-hour period preceding the interview (midnight to midnight).'",
    "nhanes_variables": [
     "DR1TKCAL",
     "DR2TKCAL"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "DR1TOT_L DR1TKCAL (DR2TOT_L DR2TKCAL); whether day 1 or the 2-day mean was used is not stated."
   },
   {
    "name": "total daily polyunsaturated fatty acids",
    "coding": "g/day (Table 1 median [IQR]); model coding not stated",
    "nhanes_variables": [
     "DR1TPFAT",
     "DR2TPFAT"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "DR1TOT_L DR1TPFAT (DR2TOT_L DR2TPFAT)."
   }
  ],
  "design": {
   "weights": "sample weights were 'considered' and Table 1 estimates 'accounted for sample weights and complex survey designs'; which weight (e.g. dietary 2-day WTDR2D or MEC) and the 6-year combination are not stated (Table 1's weighted sample size of 44,737,249 suggests weights scaled to one population)",
   "strata_psu": "'complex survey designs' accounted for in Table 1 (variables not named); not restated for the regressions",
   "quote": "To ensure the representativeness of the sample, NHANES oversamples certain subgroups of the population, so, when conducting the data analysis, we considered sample weights to correct for differential selection probabilities, to compensate for possible inadequacies in the eligible population, and to adjust for non-coverage and non-response. [...] a All estimates accounted for sample weights and complex survey designs, and percentages and means were adjusted for survey weights of NHANES. [...] Statistical analysis was performed using the R software (Version 4.2.2).",
   "software": "R software (Version 4.2.2)",
   "missing_data": "complete case for dietary data, ASCVD and income-to-poverty ratio (missing >=10%); random forest imputation for covariates with <10% missing",
   "quote_missing": "Participants with missing dietary (n = 544) or ASCVD [including coronary heart disease (CHD), angina, heart attack, and stroke, n = 34] information were excluded from the analysis, and covariates with missing values <10% were imputed using the random forest method. However, those with missing values ≥10%, such as the ratio of family income to poverty, were not imputed, and participants (n = 370) with missing information on this variable were further excluded."
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "We employed three logistic regression models to explore how CDAI relates to ASCVD and hard criteria in postmenopausal women. Model A did not adjust for any covariates; Model B made adjustments for age, race, education level, marital status, the ratio of family income to poverty, and BMI; Model C further adjusted for waist circumference, alcohol use, smoking—cigarette use, moderate to vigorous recreational activities, sleep disorders, hypertension, diabetes, family history of heart attack, NLR, HDL-C, total cholesterol, total daily caloric intake, and total daily polyunsaturated fatty acids based on Model B."
  },
  "unstated": [
   "the SD of CDAI used for the per-SD estimate (and whether the summed CDAI was standardized again)",
   "which carotenoids made up 'carotenoids' (alpha-carotene, beta-carotene, beta-cryptoxanthin, lycopene, lutein+zeaxanthin, or a subset) and whether vitamin A was RAE",
   "which mean and SD standardized each nutrient (analytic sample or other; weighted or not)",
   "the high vs low CDAI cut-off",
   "which survey weight was used and whether strata and PSU entered the regressions",
   "coding of age, income-to-poverty ratio, BMI, waist circumference, NLR, HDL-C, total cholesterol, caloric intake and PUFA in the models (Table S1 gives categories for some)",
   "how 'no menstrual information' was defined, and whether women with RHD043 = hysterectomy or other reasons under age 55 were excluded",
   "how alcohol use was coded in 2017-2018, when the '12 drinks in any one year' item was not asked",
   "whether women with only one dietary recall were excluded (the 544 with 'missing dietary' information)",
   "random forest imputation details (software, variables used)",
   "reference categories of categorical covariates"
  ],
  "notes": "The abstract's per-SD OR (0.67, 0.51-0.88) is Table 2's Model C ASCVD estimate. The Results text gives Model B per-SD as 'OR = 0.71, 95% CI: 0.58–0.81' while Table 2 prints 0.71 (0.58, 0.87). Table 1's HDL-C row ('1.00 [0.00, 1.00]') suggests HDL-C entered as a binary indicator despite its mg/dL label, and its BMI category counts (18.5-24.9: 1297 (25.4); 25.0-29.9: 356 (29.7)) look misprinted. Population: women classified by the 2013-2018 RHD043 answer or age >=55 without menstrual information (4057), observed ages 40 to 80. Covariates in Model C that cannot be rebuilt from 2021-2023 files: marital status in 6 groups (DMDMARTZ has 3), sleep disorders (no SLQ050), family history of heart attack (no MCQ300A), and alcohol use as '12 drinks in any one year' (ALQ101 absent, as it already was in 2017-2018). Weighting: the Data Source says sample weights were considered in the analysis; the regression section does not restate it. Supplementary Tables S1-S2 were fetched from Europe PMC to scratchpad/dl/supp/row319/ (nested zip; the PDF's Table S1 description column was converted to supp_s1_desc.txt). NHANES variable names are our mapping; the paper names none.",
  "adjudication": null
 },
 {
  "id": "row058",
  "rank": 113,
  "row": 58,
  "doi": "10.3389/fpsyt.2023.1274648",
  "pmcid": "PMC10623352",
  "title": "Association of non-HDL-C and depression: a cross-sectional analysis of the NHANES data",
  "authors": [
   "Zhu, Xianlin",
   "Zhao, Yiwen",
   "Li, Lu",
   "Liu, Jiaoying",
   "Huang, Qiankun",
   "Wang, Suhong",
   "Shu, Yanping"
  ],
  "year": 2023,
  "journal": "Frontiers in Psychiatry",
  "table_a": {
   "predictor": "Non-high-density lipoprotein cholesterol",
   "condition": "Depression",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "There was a positive association between non-HDL-C and depression with a 95% OR of 1.22 adjusted for multifactorial (95% CI,1.03–1.45).",
   "table_location": "Table 2 (Association between non-HDL-C and depression), row 'Q4' (reference 'Q1 | Ref'), column 'Model 3'",
   "table_quote": "|  | Unadjusted model | Model 1 | Model 2 | Model 3 | [...] | Q4 | 1.30 (1.14,1.50) | 1.40 (1.22,1.62) | 1.13 (0.96,1.33) | 1.22 (1.03,1.45) |",
   "measure": "OR",
   "estimate": 1.22,
   "ci_low": 1.03,
   "ci_high": 1.45,
   "p_value": "not printed for the estimate (Table 2, Model 3 p for trend 0.07)",
   "exposure_contrast": "highest vs lowest quartile of non-HDL-C (Q4 vs Q1, Q1 = reference); non-HDL-C = total cholesterol minus HDL-C in mmol/L; quartile cutpoints not reported",
   "model_label": "Model 3",
   "covariates_in_this_model": [
    "age",
    "sex",
    "race",
    "marital status",
    "PIR",
    "education level",
    "BMI",
    "smoking status",
    "alcohol consumption",
    "CVD",
    "diabete",
    "HDL",
    "hypertension"
   ],
   "n_analytic": 28041,
   "n_quote": "After excluding missing data for depression, non-HDL-C, and covariates, the remaining number of participants was 28,041.",
   "events": 2419
  },
  "cycles": [
   "2005-2006",
   "2007-2008",
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=18",
   "defining": "US adults aged 18 or older (both sexes; no condition named)",
   "inclusion": "NHANES 2005-2018 participants aged 18 or older (42,143) with complete PHQ-9, total cholesterol and HDL-C, and covariates.",
   "exclusions": [
    "age < 18 (Figure 1 prints 'Remove those with age< 18(N=5,575)', although 70,190 - 42,143 = 28,047)",
    "PHQ-9 missing values (N=5,746) (Figure 1)",
    "Non-HDL-C missing values (N=1,812) (Figure 1)",
    "missing values of covariant [sic] (N=6,544) (Figure 1)"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "The exclusion criteria consisted of patients with missing total cholesterol (TC) and high-density lipoprotein cholesterol (HDL-C) data, as well as incomplete responses to the Patient Health Questionnaire-9 (PHQ-9). Additionally, individuals under the age of 18 were also excluded. [...] The NHANES 2005–2018 sample consisted of a total of 42,143 adult participants. After excluding missing data for depression, non-HDL-C, and covariates, the remaining number of participants was 28,041."
  },
  "exposure": {
   "definition": "Non-HDL-C = total cholesterol minus HDL cholesterol (mmol/L), from the standard lipid panel; analyzed in quartiles with Q1 as reference. The Methods say it is calculated 'on the basis of fasting subjects’ standard lipids', but the counts in Figure 1 (34,585 adults with PHQ-9 and non-HDL-C out of 42,143) are far above the size of NHANES's morning fasting subsample (about half of examinees), so no fasting restriction appears to have been applied (our inference).",
   "nhanes_variables": [
    "LBXTC",
    "LBDTCSI",
    "LBDHDD",
    "LBDHDDSI"
   ],
   "nhanes_files": [
    "TCHOL",
    "HDL"
   ],
   "transform": "quartiles (cutpoints not reported); no continuous analysis",
   "categories": "Quartiles Q1 (reference) to Q4; cutpoints not reported (Table 2 rows are labeled Q1-Q4 only). Table 1 weighted mean: '| Non-HDL-C (mmol/L) | 3.64 ± 0.01 | 3.64 ± 0.01 | 3.73 ± 0.03 | < 0.01 |'.",
   "quote": "Non-HDL-C is calculated on the basis of fasting subjects’ standard lipids, which are total cholesterol (TC) minus high-density lipoprotein cholesterol (HDL-C)."
  },
  "outcome": {
   "definition": "Depression ('major depression'): PHQ-9 total score >= 10.",
   "nhanes_variables": [
    "DPQ010",
    "DPQ020",
    "DPQ030",
    "DPQ040",
    "DPQ050",
    "DPQ060",
    "DPQ070",
    "DPQ080",
    "DPQ090"
   ],
   "nhanes_files": [
    "DPQ"
   ],
   "quote": "In the NHANES dataset from 2005–06 to 2017–18, depression was assessed using the Patient Health Questionnaire-9 (PHQ-9). Participants were interviewed at a mobile examination center to evaluate the frequency of depressive symptoms over the past two weeks. For the purposes of this study, we defined major depression as a PHQ-9 score of 10 or higher (32)."
  },
  "covariates": [
   {
    "name": "age",
    "coding": "not stated (Table 1 gives a weighted mean)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L RIDAGEYR."
   },
   {
    "name": "sex",
    "coding": "female/male",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L RIAGENDR."
   },
   {
    "name": "race",
    "coding": "non-Hispanic white, Mexican American, non-Hispanic black, or other races",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L RIDRETH1 (Other Hispanic presumably in 'other races')."
   },
   {
    "name": "marital status",
    "coding": "married or cohabiting, never married, widowed, or divorced or separated",
    "nhanes_variables": [
     "DMDMARTL"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": false,
    "note": "DEMO_L releases only DMDMARTZ (1 Married/Living with partner, 2 Widowed/Divorced/Separated, 3 Never married), so widowed cannot be separated from divorced or separated."
   },
   {
    "name": "PIR",
    "coding": "not stated (Table 1 gives a weighted mean, so presumably continuous)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L INDFMPIR."
   },
   {
    "name": "education level",
    "coding": "college graduate or higher, some college or associate's degree, high school graduate or equivalent, or less than high school",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DEMO_L DMDEDUC2 (ages 20+; no DMDEDUC3, so 18-19-year-olds have no education value in 2021-2023)."
   },
   {
    "name": "BMI",
    "coding": "not stated for the models (Table 1 mean; Table 3 subgroups: underweight < 18.5, normal 18.5-<25, overweight 25-<30, obese >= 30)",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "BMX_L BMXBMI."
   },
   {
    "name": "smoking status",
    "coding": "never, former, or current",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "Definitions not stated; SMQ_L SMQ020, SMQ040."
   },
   {
    "name": "alcohol consumption",
    "coding": "never, former, or current",
    "nhanes_variables": [
     "ALQ101",
     "ALQ110",
     "ALQ120Q",
     "ALQ111",
     "ALQ121"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "Definitions not stated; the ALQ items changed in 2017-2018 (ALQ111 ever drank, ALQ121 past-12-month frequency), which ALQ_L keeps, so never/former/current can be built from ALQ111 and ALQ121."
   },
   {
    "name": "CVD",
    "coding": "yes/no; history of heart failure, angina, myocardial infarction, and stroke",
    "nhanes_variables": [
     "MCQ160B",
     "MCQ160D",
     "MCQ160E",
     "MCQ160F"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "MCQ_L MCQ160b, MCQ160d, MCQ160e, MCQ160f (ages 20+)."
   },
   {
    "name": "diabetes ('diabete' in the Table 2 note)",
    "coding": "yes/no; definition not stated",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "Definition not stated; DIQ_L DIQ010 (and DIQ050/DIQ070, GHB_L, GLU_L if a lab-based definition was used)."
   },
   {
    "name": "HDL",
    "coding": "not stated (Table 1 gives HDL-C in mmol/L); in the Table 2 note only",
    "nhanes_variables": [
     "LBDHDD",
     "LBDHDDSI"
    ],
    "nhanes_files": [
     "HDL"
    ],
    "in_2021_2023": true,
    "note": "HDL_L LBDHDD. Listed for Model 3 in the Table 2 note but not in the Methods' Model 3 description; it is also a component of the exposure."
   },
   {
    "name": "hypertension",
    "coding": "yes/no; definition not stated",
    "nhanes_variables": [
     "BPQ020"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Definition not stated; BPQ_L BPQ020, BPQ150, and oscillometric BP (BPXO_L) exist in 2021-2023; earlier cycles used auscultatory BP (BPX)."
   }
  ],
  "design": {
   "weights": "'appropriate 14-year sample weights' constructed per NHANES recommendations (which 2-year weight was combined is not stated)",
   "strata_psu": "not stated",
   "quote": "The statistical analysis was conducted using R version 4.3.0. To ensure accurate results, appropriate 14-year sample weights were constructed following NHANES recommendations.",
   "software": "R version 4.3.0",
   "missing_data": "complete case (missing PHQ-9, non-HDL-C, and covariates excluded; Figure 1)",
   "quote_missing": "After excluding missing data for depression, non-HDL-C, and covariates, the remaining number of participants was 28,041."
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "We examined the association between non-HDL-C and depression using weighted multivariable logistic regression models and subgroup analysis. [...] Initially, logistic regression was employed to calculate the ratio of the 95% confidence interval between depression and non-HDL-C, with the first quartile serving as the reference."
  },
  "unstated": [
   "quartile cutpoints of non-HDL-C and whether they were computed weighted or unweighted",
   "which 2-year weight was combined into the 14-year weight (MEC or fasting subsample), and whether strata and PSU were used",
   "whether the analysis was restricted to fasting participants, as the exposure definition's wording suggests",
   "definitions of diabetes and hypertension",
   "definitions of never/former/current smoking and alcohol use",
   "how age, PIR and BMI entered the models (continuous or categorical)",
   "whether HDL was in Model 3 (Table 2 note: yes; Methods: not listed) and its coding",
   "reference categories of categorical covariates",
   "handling of partially completed PHQ-9 questionnaires",
   "how 18-19-year-olds got an education level and CVD history, which NHANES asks of ages 20+ (they may have been dropped as missing covariates)"
  ],
  "notes": "The abstract states the headline as 'a 95% OR of 1.22 adjusted for multifactorial (95% CI,1.03–1.45)' without naming the contrast; Table 2 and the Results identify it as Model 3, highest vs lowest quartile. Other inconsistencies: the abstract says 'There were 42,143 participants in this study' (that is the adult count before exclusions; the analysis has 28,041) and gives the weighted means in reversed order ('3.64 vs. 3.73'); the abstract's subgroup ORs (normal BMI 0.93 (0.66-1.32); no hypertension 1.29 (1.01-1.66)) do not match Table 3 (normal BMI Q4 1.39 (0.92,2.11); no hypertension Q4 1.30 (1.01,1.68)); the Results write 'quintile' for quartiles; the sensitivity text (IPTW 1.18 (1.04-1.33); excluding antidepressants 1.27 (1.03-1.57)) differs from Table 4 (1.18 (1.04,1.35); 1.28 (1.03,1.58)); Figure 1's age exclusion (N=5,575) does not fit its own boxes (70,190 to 42,143). The Model 2 Q4 estimate is not significant (1.13, 0.96-1.33) and the Model 3 trend is not (p = 0.07). Model 3 per the Table 2 note adjusts for HDL, a component of the exposure. Figure 1 (flow chart) is an image; its counts were transcribed from the publisher's figure file, saved to scratchpad/dl/supp/row058/fig1.jpg; the paper has no supplement. NHANES variable names are our mapping; the paper names none.",
  "adjudication": null
 },
 {
  "id": "row072",
  "rank": 115,
  "row": 72,
  "doi": "10.1097/MD.0000000000037315",
  "pmcid": "PMC10919533",
  "title": "A positive association between RDW and coronary heart disease in the rheumatoid arthritis population: A cross-sectional study from NHANES",
  "authors": [
   "Zhang, Mei Qi",
   "Tan, Wen Tao",
   "Li, Wei Dong",
   "Shen, Xuan Yang",
   "Shen, Yuan",
   "Jiang, Xiao Lu",
   "Wen, Hong Fu"
  ],
  "year": 2024,
  "journal": "Medicine",
  "table_a": {
   "predictor": "Red blood cell distribution width",
   "condition": "Coronary heart disease",
   "population": "US adults with rheumatoid arthritis"
  },
  "headline": {
   "abstract_quote": "RDW and coronary heart disease were found to have a positive association in the rheumatoid arthritis population (OR = 1.145, 95%CI: 1.036–1.266, P = .0098), even after adjusting for factors such as age, gender, race, education level, smoking, and drinking.",
   "table_location": "Table 3 (Multiple logistic regression analysis of the relationship between red blood cell distribution width and coronary heart disease among the rheumatoid arthritis population), row 'RDW', columns 'Model 1' (OR, 95%CI, P). The same values are in Table 2 (Univariate logistic regression analysis), row 'RDW'. The abstract text is from the Europe PMC XML; the text conversion dropped the abstract body.",
   "table_quote": "| RDW | 1.145 | (1.036, 1.266) | .0098 | 1.168 | (1.060, 1.287) | .0026 | 1.187 | (1.065, 1.322) | .0029 |",
   "measure": "OR",
   "estimate": 1.145,
   "ci_low": 1.036,
   "ci_high": 1.266,
   "p_value": ".0098",
   "exposure_contrast": "per 1-unit increase in RDW (%), RDW entered as a continuous variable (the unit of the contrast is not stated; NHANES RDW is in percent)",
   "model_label": "Model 1 (\"Model 1: we did not adjust any covariants\"; identical to the univariate Table 2 estimate). The abstract describes it as adjusted for age, gender, race, education level, smoking and drinking, which matches neither Model 2 (age, gender, race) nor Model 3 (all Table 1 covariates).",
   "covariates_in_this_model": [],
   "n_analytic": 1236,
   "n_quote": "\"Ultimately, our sample comprised 1236 individuals (Fig. 1).\" (2.1 Population); \"The total number of participants in this study was 1236 (Table 1).\" (3. Results); Figure 1 (image, transcribed): \"Participants with RA N=(1,236)\". Table 1's tertile sizes, \"| N | 402 | 408 | 425 |  |\", sum to 1,235.",
   "events": 102
  },
  "cycles": [
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-March 2020 (pre-pandemic)"
  ],
  "population": {
   "age": ">=20 (not stated; implied because the MCQ arthritis and CHD items are asked of ages 20+, and Table 4's youngest age band is '20–55')",
   "defining": "rheumatoid arthritis, self-reported physician or other health professional diagnosis",
   "inclusion": "NHANES 2011-2020 participants with RDW data, CHD data, and self-reported rheumatoid arthritis",
   "exclusions": [
    "Start: 45,462 participants in the 4 cycles",
    "No RDW data: 8,692 excluded, leaving 36,770",
    "No CHD data ('without CHD' in Figure 1, 'CHD values' missing in the text): 13,111 excluded, leaving 23,659",
    "Not RA ('without RA' in Figure 1, 'RA values' missing in the text): 22,423 excluded, leaving 1,236"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "\"A total of 45,462 participants were involved in the NHANES survey over all 4 cycles. In our study, RDW values were missing for 8692 participants, CHD values for 13,111, and RA values for 22,423. Ultimately, our sample comprised 1236 individuals (Fig. 1).\" (2.1 Population); \"RA diagnosis was based on reports from a physician or other healthcare professional.\" (2.2 Variables); Figure 1 (image, transcribed): \"Participants with RDW data(n=36,770)\", \"Participants without CHD n=(13,111)\", \"Participants with CHD (n=23,659)\", \"Participants without RA n=(22,423)\", \"Participants with RA N=(1,236)\"."
  },
  "exposure": {
   "definition": "Red blood cell distribution width (RDW, %), from the complete blood count, as a continuous variable for the headline; tertiles (low, middle, high) for Table 1 and a trend test. The paper says RDW was obtained 'using a questionnaire', which is wrong: RDW is a CBC laboratory value.",
   "nhanes_variables": [
    "LBXRDW"
   ],
   "nhanes_files": [
    "CBC"
   ],
   "transform": "none (continuous, per 1 unit); tertiles for Table 1 and the P for trend",
   "categories": "Tertiles, cutpoints not reported: \"RDW tertiles were classified as low, middle, and high.\" (2.2 Variables)",
   "quote": "\"RDW examination was conducted using a questionnaire. RDW tertiles were classified as low, middle, and high.\" (2.2 Variables); \"RDW, a routine blood test parameter calculated as the standard deviation of red cell volume divided by the mean cell volume, expressed as a percentage (RDW%), reflects the variability in red blood cell size.\" (1. Introduction)"
  },
  "outcome": {
   "definition": "Coronary heart disease, self-reported (ever told by a doctor or other health professional; presumably MCQ160C = 1 vs 2). The paper gives no definition beyond 'CHD values'.",
   "nhanes_variables": [
    "MCQ160C"
   ],
   "nhanes_files": [
    "MCQ"
   ],
   "quote": "\"In our study, RDW values were missing for 8692 participants, CHD values for 13,111, and RA values for 22,423.\" (2.1 Population); \"Both univariate and multivariate logistic regression were employed to assess the relationship between RDW and CHD.\" (2.3 Statistical analysis)"
  },
  "covariates": [
   {
    "name": "Age (yr)",
    "coding": "continuous, years (Table 1 mean ± SD; Table 2 OR per year)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Model 2 and Model 3."
   },
   {
    "name": "Gender",
    "coding": "Male (reference), Female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Model 2 and Model 3."
   },
   {
    "name": "Race",
    "coding": "Mexican American (reference), 'Hispanic White', 'Hispanic Black', Other Race (labels as printed, almost certainly non-Hispanic White and non-Hispanic Black; where Other Hispanic went is not stated)",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Model 2 and Model 3. Four levels, while RIDRETH1 has five."
   },
   {
    "name": "Poverty",
    "coding": "continuous (Table 1 mean ± SD about 2.1, so the family income-to-poverty ratio)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only."
   },
   {
    "name": "Education",
    "coding": "Less than high school (reference), High school, University and above",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only."
   },
   {
    "name": "BMI (kg/m²)",
    "coding": "continuous",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only."
   },
   {
    "name": "Hypertension",
    "coding": "Yes (reference in Table 2), No; definition not stated ('a history of ... hypertension')",
    "nhanes_variables": [
     "BPQ020"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only. BPQ020 is the likely source; the paper does not say."
   },
   {
    "name": "Smoking",
    "coding": "Yes (reference), No: smoked at least 100 cigarettes in lifetime",
    "nhanes_variables": [
     "SMQ020"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only."
   },
   {
    "name": "Drinking",
    "coding": "Yes, No, Other (Table 1; 'Other' has 243 participants and is probably missing or unknown): 'regular alcohol consumption', definition not stated",
    "nhanes_variables": [
     "ALQ101",
     "ALQ111",
     "ALQ121"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only. The item used is not stated (2011-2016 ALQ101 'at least 12 drinks in any one year'; 2017-2020 ALQ111/ALQ121); ALQ_L has ALQ111 and ALQ121."
   },
   {
    "name": "Diabetes",
    "coding": "Yes (reference), No; definition not stated ('a history of diabetes')",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only. DIQ010 is the likely source; the paper does not say."
   },
   {
    "name": "Hypercholesterolemia",
    "coding": "Yes (reference), No; definition not stated",
    "nhanes_variables": [
     "BPQ080"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3 only. BPQ080 is the likely source; the paper does not say."
   },
   {
    "name": "Physical activity (moderate recreational activities)",
    "coding": "Activity Yes (reference), No (Table 2)",
    "nhanes_variables": [
     "PAQ665"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": true,
    "note": "Listed in the Methods and Table 2 but not in Table 1, and Model 3 is 'all covariants in Table 1', so it may not be in Model 3. 2021-2023 PAQ_L asks the frequency of moderate leisure-time activity (PAD790Q/PAD790U), from which a yes/no can be built; the 2011-2020 item PAQ665 asks yes/no directly."
   }
  ],
  "design": {
   "weights": "unstated: 'a weighted approach' only (no weight variable named, and nothing on combining 2011-2016 two-year weights with the 2017-March 2020 pre-pandemic weights)",
   "strata_psu": "unstated",
   "quote": "\"Data analysis was performed using a weighted approach.\" (2.3 Statistical analysis)",
   "software": "R version 4.0.4",
   "missing_data": "unstated; participants missing RDW, CHD or RA were excluded, and Table 1 keeps an 'Other' level for Drinking, suggesting a missing category for that covariate",
   "quote_missing": "\"In our study, RDW values were missing for 8692 participants, CHD values for 13,111, and RA values for 22,423.\" (2.1 Population); Table 1, Drinking: \"| Other | 75 (18.657%) | 85 (20.833%) | 83 (19.529%) |  |\""
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "\"Data analysis was performed using a weighted approach. Both univariate and multivariate logistic regression were employed to assess the relationship between RDW and CHD. A general additive model and subgroup analysis were also utilized. R version 4.0.4 was used for all statistical analyses.\" (2.3 Statistical analysis)"
  },
  "unstated": [
   "Which survey weight was used and how the 2011-2016 two-year weights were combined with the 2017-March 2020 pre-pandemic weights ('a weighted approach' only)",
   "Whether strata and PSUs were used for variance estimation",
   "The cycles by name (only 'between 2011 and 2020' and '4 cycles'; the 2017-March 2020 pre-pandemic file is inferred from the 45,462 total)",
   "How RA was coded (presumably MCQ160A = 1 and MCQ195 = 2) and whether 'refused' or 'don't know' answers were excluded or counted as not RA",
   "How CHD was coded (presumably MCQ160C = 1 vs 2) and the handling of 'refused' or 'don't know'",
   "Any age restriction (none stated; the MCQ items cover ages 20+)",
   "The unit of the RDW contrast (per 1 percentage point assumed)",
   "RDW tertile cutpoints",
   "Definitions of hypertension, diabetes, hypercholesterolemia and drinking",
   "Whether physical activity is in Model 3 (Methods and Table 2 include it; Table 1, which defines Model 3, does not)",
   "Handling of missing covariate values (Table 1 shows an 'Other' Drinking level)",
   "Coding of age, poverty and BMI in Models 2 and 3 (Table 1 treats them as continuous)",
   "Reference levels in Model 3 (Table 2 uses Male, Mexican American, Less than high school, and 'Yes' for the yes/no covariates)"
  ],
  "notes": "1) The abstract's estimate (OR 1.145, 1.036 to 1.266, P = .0098) is Table 3 Model 1 (no adjustment) and Table 2 (univariate), but the abstract says it holds 'even after adjusting for factors such as age, gender, race, education level, smoking, and drinking'. By the headline rule it is the headline; the fully adjusted Model 3 estimate is OR 1.187 (1.065, 1.322), P .0029. 2) Cycles: 45,462 matches 2011-2012 + 2013-2014 + 2015-2016 + 2017-March 2020 pre-pandemic (P_ files), so the 2017-2018 files were not combined with the overlapping pre-pandemic files; the 2017-2020 data need P_CBC, P_MCQ, P_DEMO and so on, and the pre-pandemic MEC weight (WTMECPRP) if weights are combined. 3) Race labels 'Hispanic White' and 'Hispanic Black' are almost certainly non-Hispanic White and non-Hispanic Black. 4) The Methods say 'RDW examination was conducted using a questionnaire', which is an error (CBC). 5) Other internal inconsistencies: Table 1 tertile Ns sum to 1,235 vs 1,236; the abstract's age band '55–66' vs Table 4's '56–66'; the text gives the male subgroup 'P = .084' vs Table 4 '.0184'; Table 2's poverty OR 1.205 (1.001, 1.450) has P .0529 although its CI excludes 1; the Results say BMI is associated with CHD although Table 2 gives P .4009. 6) The sample flow is sequential (Figure 1): 45,462 to 36,770 with RDW, to 23,659 with CHD data, to 1,236 with RA. 7) Sources read: work/fulltext_txt/row072_PMC10919533.txt, the abstract from dl/fulltext/row072_PMC10919533.xml (saved as dl/supp/row072/abstract_from_xml.txt), and Figure 1 (dl/supp/row072/g001.jpg, transcribed in dl/supp/row072/fig1_transcription.txt).",
  "adjudication": null
 },
 {
  "id": "row249",
  "rank": 126,
  "row": 249,
  "doi": "10.1186/s40795-024-00938-7",
  "pmcid": "PMC11437793",
  "title": "The association between adult asthma in the United States and dietary total energy intake: a retrospective cross-sectional analysis from NHANES",
  "authors": [
   "Cao, Xianghua",
   "Lu, Tong",
   "Tu, Yunyun",
   "Zhou, Rongguan",
   "Li, Xueping",
   "Du, Linjun"
  ],
  "year": 2024,
  "journal": "BMC Nutrition",
  "table_a": {
   "predictor": "Dietary total energy intake",
   "condition": "Asthma",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "After adjusting for confounders, odds ratios (OR) for asthma decreased with higher energy intake: Q2 (OR = 0.77, 95% CI: 0.69–0.86, p < .001), Q3 (OR = 0.66, 95% CI: 0.59–0.75, p < .001), and Q4 (OR = 0.61, 95% CI: 0.53–0.69, p < .001) compared to Q1 (< 17.73 kcal/kg/day).",
   "table_location": "Table 2 (Association between dietary total energy intake (kcal/kg/day) and asthma), row 'Q4', column 'Model 3' (OR (95%CI)) and the p-value column after it",
   "table_quote": "| Q4 | 5339 | 735 (13.8) | 0.74 (0.67 ~ 0.83) | < 0.001 | 0.77 (0.69 ~ 0.86) | < 0.001 | 0.76 (0.68 ~ 0.85) | < 0.001 | 0.61(0.53 ~ 0.69) | < 0.001 |",
   "measure": "OR",
   "estimate": 0.61,
   "ci_low": 0.53,
   "ci_high": 0.69,
   "p_value": "< 0.001",
   "exposure_contrast": "Q4 (> 33.29 kcal/kg/day) vs Q1 (< 17.73 kcal/kg/day, reference) of total energy intake per kg body weight; the abstract lists Q2, Q3 and Q4 vs Q1, and the quartile coding is represented by its highest category",
   "model_label": "Model 3",
   "covariates_in_this_model": [
    "gender",
    "age",
    "race and ethnicity",
    "familial asthma",
    "PIR",
    "white blood cell count",
    "percentage of eosinophils",
    "hemoglobin count",
    "Vitamin D",
    "n-3 PUFAs",
    "n-6 PUFAs",
    "total fiber",
    "Zn"
   ],
   "n_analytic": 21354,
   "n_quote": "[Results] As a result, 3154 (14.77%) of the 21,354 participants developed asthma in the final analysis (Fig. 1). [Table 2, column 'No.'] | Q1 | 5339 | ... | Q4 | 5339 | ... | Trend. test | 21,354 |",
   "events": 3154
  },
  "cycles": [
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=20",
   "defining": "US adults (aged 20 or older); no other defining characteristic",
   "inclusion": "NHANES 2009-2018 participants aged 20 or older with an answer to the asthma question, dietary total energy data, and complete covariate data",
   "exclusions": [
    "Younger than 20 years (n = 20,858); 28,835 remain (Fig. 1)",
    "Missing asthma questionnaire (n = 25)",
    "Missing total dietary energy (n = 3,463) (the Methods give the two together as n = 3488)",
    "Missing covariates (n = 3,993 in Results and Fig. 1; n = 3981 in Methods, Covariates); 21,354 analyzed"
   ],
   "exclusions_not_in_2021_2023": [
    "Missing familial asthma (part of the missing-covariate step): MCQ_L 2021-2023 has no family-history-of-asthma item (MCQ300B absent)"
   ],
   "quote": "[Methods, The study population and the data source] A total of 49,693 individuals finished the interviews. Individuals younger than 20 years old (n = 20858) were excluded. Additionally, individuals (n = 3488) who lacked complete information about their dietary intake of total energy intake and asthma were also not included. [Methods, Covariates] Covariates with missing data were not included (n = 3981). Lastly, 21,354 people were included in the analysis. [Results] We eliminated those under the age of twenty (20,858), those without completed asthma questionnaires (n = 25), and those lacking complete dietary energy information (n = 3463). Data with missing variables were also excluded (n = 3993). [Fig. 1, image transcribed] 2009-2018 Participants who completed the interview (n=49693) ... Participants who less than 20 years old (n=20858) ... Participants who more than 20 years old (n=28835) ... Excluded data with missing ashma (n=25) ... Excluded data with missing total dietary energy (n=3463) ... Excluded data with missing covariates (n=3993) ... Participants included in analysis (n=21354)"
  },
  "exposure": {
   "definition": "Dietary total energy intake per kg body weight (kcal/kg/day). The Methods describe the mean of the two 24-h recalls divided by body weight. The authors' dataset (Supplementary Material 2) holds only the day-1 total (column 'kcal', integers, i.e. DR1TKCAL) and body weight ('Wt', BMXWT), and day-1 kcal divided by weight reproduces the paper's quartile cutpoints and Table 2 (see notes). Analyzed in quartiles, Q1 reference.",
   "nhanes_variables": [
    "DR1TKCAL",
    "DR2TKCAL (named by the Methods' two-recall mean; absent from the authors' dataset)",
    "BMXWT"
   ],
   "nhanes_files": [
    "DR1TOT_F",
    "DR1TOT_G",
    "DR1TOT_H",
    "DR1TOT_I",
    "DR1TOT_J",
    "DR2TOT_F to DR2TOT_J",
    "BMX_F to BMX_J",
    "DR1TOT_L, DR2TOT_L and BMX_L (2021-2023)"
   ],
   "transform": "kcal/day divided by body weight (kg); quartiles",
   "categories": "[Methods, Dietary total energy intake] Finally, participants were divided into four groups based on total energy intake (kcal/kg/day) quartile amounts: Q1 (<17.73 kcal/kg/day; n = 5339), Q2 (17.73–24.52 kcal/kg/day; n = 5338), Q3 (21.52–33.29 kcal/kg/day; n = 5338), Q4 (>33.29 kcal/kg/day; n = 5339).",
   "quote": "[Methods, Dietary total energy intake] 24-hour dietary recall interview were performed twice in order to assess the dietary intake of total energy intake. The interviews were conducted face-to-face at the MEC ... The average daily consumption (kcal/kg/day) ... of total energy intake was computed by averaging the results of two interviews, adjusting for participant weight, and excluding those with incomplete participant data."
  },
  "outcome": {
   "definition": "Self-reported ever doctor-diagnosed asthma: MCQ010 = 1 (yes) vs 2 (no); refused/don't know excluded as missing",
   "nhanes_variables": [
    "MCQ010"
   ],
   "nhanes_files": [
    "MCQ_F",
    "MCQ_G",
    "MCQ_H",
    "MCQ_I",
    "MCQ_J",
    "MCQ_L (2021-2023)"
   ],
   "quote": "[Methods, Asthma] In the NHANES database, we defined asthma status based on responses to the question [7] “Has a doctor or other health professional ever told you that you have asthma?” A response of “Yes” indicates asthma, while a response of “No” indicates non-asthma."
  },
  "covariates": [
   {
    "name": "gender",
    "coding": "male/female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Model 3"
   },
   {
    "name": "age",
    "coding": "years, continuous (Table 1 mean ± SD)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Model 3"
   },
   {
    "name": "race and ethnicity",
    "coding": "Mexican American, Other Hispanic, Non-Hispanic White, Non-Hispanic Black, Other/Multiracial (Table 1); the Methods list only the first four",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Model 3. The authors' dataset column 'TH1' has codes 1-5, matching RIDRETH1; reference category unstated (Table S1 uses Mexican American as reference)."
   },
   {
    "name": "PIR",
    "coding": "low (PIR <= 1.3), medium (> 1.3 to 3.5), high (> 3.5)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Model 3; refitting with these three categories reproduces Table 2"
   },
   {
    "name": "familial asthma",
    "coding": "yes/no",
    "nhanes_variables": [
     "MCQ300B"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": false,
    "note": "In Model 3. MCQ300B (close blood relative ever told had asthma) is in the authors' dataset; no family-history item in MCQ_L 2021-2023. Refused/don't know (7/9) presumably dropped: keeping only 1/2 brings the sample to within 12 of 21,354."
   },
   {
    "name": "white blood cell count",
    "coding": "continuous, 1000 cells/uL",
    "nhanes_variables": [
     "LBXWBCSI"
    ],
    "nhanes_files": [
     "CBC"
    ],
    "in_2021_2023": true,
    "note": "In Model 3"
   },
   {
    "name": "percentage of eosinophils",
    "coding": "continuous, %",
    "nhanes_variables": [
     "LBXEOPCT"
    ],
    "nhanes_files": [
     "CBC"
    ],
    "in_2021_2023": true,
    "note": "In Model 3"
   },
   {
    "name": "hemoglobin count",
    "coding": "continuous, g/dL",
    "nhanes_variables": [
     "LBXHGB"
    ],
    "nhanes_files": [
     "CBC"
    ],
    "in_2021_2023": true,
    "note": "In Model 3"
   },
   {
    "name": "Vitamin D",
    "coding": "continuous serum 25OHD, nmol/L (Table 1 median 62.5)",
    "nhanes_variables": [
     "LBXVIDMS"
    ],
    "nhanes_files": [
     "VID"
    ],
    "in_2021_2023": true,
    "note": "In Model 3; 2021-2023 VID_L LBXVIDMS"
   },
   {
    "name": "n-3 PUFAs",
    "coding": "continuous; the authors' dataset column 'n3' equals DR1TP183 + DR1TP184 + DR1TP205 + DR1TP225 + DR1TP226 in g/day (Table 1 labels it mg/kg/day but its median 1.5 is g/day)",
    "nhanes_variables": [
     "DR1TP183",
     "DR1TP184",
     "DR1TP205",
     "DR1TP225",
     "DR1TP226"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3; g/day reproduces Table 2 better than mg/kg/day"
   },
   {
    "name": "n-6 PUFAs",
    "coding": "continuous; components unstated (Table 1 median 14.4, g/day despite the mg/kg/day label)",
    "nhanes_variables": [
     "DR1TP182 and DR1TP204 (presumed; the dataset gives only the total 'n6')"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3"
   },
   {
    "name": "total fiber",
    "coding": "continuous, g/day",
    "nhanes_variables": [
     "DR1TFIBE"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3"
   },
   {
    "name": "Zn",
    "coding": "continuous dietary zinc, mg/day (Table 1 median 9.6)",
    "nhanes_variables": [
     "DR1TZINC"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3"
   },
   {
    "name": "physical exercise",
    "coding": "sedentary, moderate, vigorous (described with 30-day vigorous/moderate items)",
    "nhanes_variables": [
     "unstated (the 30-day wording matches 1999-2006 PAD200/PAD320; 2009-2018 has GPAQ PAQ650/PAQ665)"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": true,
    "note": "Not in Model 3 per the Table 2 note (in Model 1, in Model 3 per the Methods, and in the Table 3 adjustment). 2021-2023 PAQ_L has leisure-time moderate and vigorous frequency (PAD790Q/U, PAD810Q/U), enough for a sedentary/moderate/vigorous split but not the same items. Absent from the authors' dataset."
   },
   {
    "name": "smoking status",
    "coding": "never (< 100 cigarettes), current, former",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "Not in Model 3 per the Table 2 note (in Model 1, in Model 3 per the Methods, and in the Table 3 adjustment). Absent from the authors' dataset."
   },
   {
    "name": "BMI",
    "coding": "< 25, 25-30, >= 30",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Described among covariates but in no model; requiring non-missing BMI brings the sample closest to 21,354 (see notes)"
   }
  ],
  "design": {
   "weights": "none: the paper never mentions sample weights; an unweighted logistic regression on the authors' dataset reproduces Table 2 (see notes)",
   "strata_psu": "not used (never mentioned)",
   "quote": "[Methods, Statistical analysis] The statistical programs IBM SPSS, version 25.0 (IBMCorp., Armonk, N.Y., USA), and Free Statistics, version 1.9, were used for all analyses. The independent and chi-squared tests were carried out to examine the differences in continuous and categorical variables, respectively.",
   "software": "IBM SPSS 25.0 and Free Statistics 1.9",
   "missing_data": "complete case",
   "quote_missing": "[Methods, Covariates] Covariates with missing data were not included (n = 3981). [Results] Data with missing variables were also excluded (n = 3993)."
  },
  "model": {
   "family": "logistic",
   "weighted": false,
   "quote": "[Methods, Statistical analysis] We employed these logistic regression models to ascertain the association between asthma and total energy consumption (kcal/kg/day). Three adjustment models were constructed for the multivariate logistic regression analysis"
  },
  "unstated": [
   "Sample weights, strata and PSUs (none mentioned; the analysis was evidently unweighted)",
   "Which recall(s) define energy intake: the Methods say the mean of two recalls, but the authors' dataset holds only the day-1 total, and day-1 energy reproduces the results",
   "Whether recalls not flagged reliable (DR1DRSTZ not 1) were excluded",
   "The n-6 PUFA components (presumably DR1TP182 + DR1TP204) and the units entered in the model (Table 1 says mg/kg/day; the values are g/day, and g/day reproduces Table 2)",
   "Which covariates Model 3 really holds: the Methods include physical exercise and smoking status, the Table 2 note does not (the authors' dataset has neither, and the Table 2 note's list reproduces the estimates)",
   "How physical exercise was built from NHANES items (the 30-day wording does not match the 2009-2018 GPAQ)",
   "How refused/don't know answers on familial asthma were handled",
   "How continuous covariates entered the model (presumably linear) and the race reference category",
   "How quartile cutpoints were computed (sample quartiles, unweighted) and which side of each boundary is inclusive",
   "Whether pregnant women were excluded",
   "Exact p-value of the headline estimate (printed '< 0.001')"
  ],
  "notes": "Headline rule: the abstract's first exposure coding is energy quartiles (reference Q1, the lowest), so per the clarified rule it is represented by Q4 vs Q1, Model 3 OR 0.61 (0.53-0.69), p < 0.001, though the abstract lists Q2 (0.77, 0.69-0.86) first. Reproduction check (allowed cycles only): Supplementary Material 2 (40795_2024_938_MOESM2_ESM.csv, saved under scratchpad/dl/supp/row249/files/) is the authors' dataset for all 49,693 participants of 2009-2018 (columns SEQN, kcal, fibe, zn, DR1TP183-DR1TP226, n3, n6, gender, age, TH1, PIR, Wt, BMI, MCQ010, MCQ300B, WBC, EOPCT, Hg, VD and others; no smoking or physical activity). 'kcal' is integer-valued with 7,339 missing, i.e. day-1 DR1TKCAL, not a two-day mean. Taking age >= 20, MCQ010 in 1/2, non-missing kcal, MCQ300B in 1/2 and complete gender, age, TH1, PIR, WBC, EOPCT, Hg, VD, n3, n6, fibe, zn, Wt and BMI gives N = 21,366 with 3,156 asthma (paper 21,354 and 3,154), quartile cutpoints of kcal/Wt 17.73, 24.515, 33.295 (paper 17.73, 24.52, 33.29), and per-quartile asthma 943/776/700/737 (paper 943/775/701/735). Unweighted glm (R 4.6.1) with the Table 2 note's Model 3 covariates (PIR in 3 categories, RIDRETH1 5 categories, n-3/n-6 in g/day) gives Q2 0.769 (0.688-0.859), Q3 0.664 (0.589-0.748), Q4 0.612 (0.533-0.703) vs the paper's 0.77 (0.69-0.86), 0.66 (0.59-0.75), 0.61 (0.53-0.69); Model 2 Q2 0.821 (0.737-0.915) vs 0.82 (0.74-0.91); unadjusted Q2 0.793 (0.715-0.880) vs 0.79 (0.71-0.88). The remaining 12 exclusions are probably missing smoking or physical activity (Table 1 has no missing categories for them). Scripts: scratchpad/scripts/x_r249_133_220_275/r249_refit2.R. Text discrepancies: Q3's lower bound is printed 21.52 (should be 24.52) in the Methods, Table 2 note and Fig. 3; Methods exclusions 3488 (= 25 + 3463) and 3981 vs Results/Fig. 1 3993; females 10,349 (48.5%) in Table 1 vs '11,010 (51.5%)' in Results; non-Hispanic White 9034 vs '9,037'; Results say '16,535, 77.4% of them, were over 65' with a mean age of 49.2; the GAM threshold is written '24 mg/kg/day' (Table 3 uses 23.83 kcal/kg/day); the left-segment OR 0.959 is called a '3.1% reduction' (Results) and a '1.7% decrease' (Discussion); univariate Q4 OR is 0.74 in Table 2 but 0.75 in Table S1. In 2021-2023 both dietary recalls were by telephone; familial asthma (MCQ300B) cannot be built. The paper uses only the 2009-2010 to 2017-2018 cycles; no 2017-March 2020 pre-pandemic (P_) files, so no overlap issue.",
  "adjudication": null
 },
 {
  "id": "row133",
  "rank": 128,
  "row": 133,
  "doi": "10.1136/bmjopen-2023-082601",
  "pmcid": "PMC11407204",
  "title": "Association of sleep duration with Visceral Adiposity Index: a cross-sectional study based on the NHANES 2007–2018",
  "authors": [
   "Liu, Juan",
   "Gao, Yajie",
   "Ye, Nan",
   "He, Xingkang",
   "Zhang, Jing"
  ],
  "year": 2024,
  "journal": "BMJ Open",
  "table_a": {
   "predictor": "Sleep health",
   "condition": "Visceral adiposity index",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "After adjusting for the sociodemographic, lifestyle and other covariates, short sleep was significantly linked to increased VAI (β=0.15, 95% CI 0.01 to 0.28) in relation to middle sleep duration, whereas no significant association was found between long sleep duration and VAI.",
   "table_location": "Table 2 (Multivariate analysis of association between sleep duration and Visceral Adiposity Index), row 'Short (<7 hours/day)', column 'Model III' (β (95% CI); p value)",
   "table_quote": "| Short (<7 hours/day) | 0.20 (0.06 to 0.33); <0.01 | 0.21 (0.07 to 0.34); <0.01 | 0.15 (0.01 to 0.28); 0.04 |",
   "measure": "beta",
   "estimate": 0.15,
   "ci_low": 0.01,
   "ci_high": 0.28,
   "p_value": "0.04",
   "exposure_contrast": "short sleep (< 7 hours/day) vs middle sleep (7-9 hours/day, reference) on weekdays or workdays; difference in mean VAI (unitless index)",
   "model_label": "Model III",
   "covariates_in_this_model": [
    "age",
    "gender",
    "race",
    "marital status",
    "education",
    "poverty",
    "smoke status",
    "alcohol drinking status",
    "physical activity",
    "energy intake",
    "hypertension",
    "diabetes",
    "stroke",
    "heart attack",
    "congestive heart failure",
    "coronary heart disease",
    "cancer"
   ],
   "n_analytic": 11252,
   "n_quote": "[Abstract, Participants] A total 11 252 eligible participants who have complete information for sleep duration and VAI. [Results] A total of 11 252 participants were included in our study based on our inclusion and exclusion criteria. [Table 1 header] | <7 (n=3885) | 7–9 (n=6823) | >9 (n=544) |",
   "events": null
  },
  "cycles": [
   "2007-2008",
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=18 as stated; the analyzed ages were 20-80",
   "defining": "US adults; no other defining characteristic",
   "inclusion": "Participants aged 18 or older in the six cycles with complete sleep duration, VAI and covariate data (age, gender, race, marital status, education, poverty, smoking, drinking, physical activity, energy intake, BMI, healthy eating index, self-reported chronic diseases)",
   "exclusions": [
    "Missing VAI or sleep information (N=22,776): 38,562 aged >= 18 -> 15,786",
    "Missing age, gender, race, marriage, education, poverty, smoke, alcohol use, physical activity, body mass index, healthy eating index or self-reported diseases (N=4,534) -> 11,252"
   ],
   "exclusions_not_in_2021_2023": [
    "Missing healthy eating index: HEI-2015 needs USDA food pattern equivalents, outside NHANES (requiring a complete day-1 recall approximates the missingness)",
    "Missing physical activity as measured in 2007-2018 (GPAQ work, transport and leisure items); PAQ_L 2021-2023 has leisure-time items only"
   ],
   "quote": "[Methods, Study population] In the present study, participants ≥18 years of age who were included in the study had to provide complete information regarding sleep duration, VAI, age, gender, race, marital status, education status, poverty, smoking status, drinking status, physical activity, energy intake and self-reported chronic diseases including hypertension, stroke, heart attack, congestive heart failure, coronary heart disease, diabetes and cancer. Participants who had missing data for VAI, sleep duration or other important covariates at the baseline assessment were excluded. [Results] The average age of the participants was 49 years, ranging from 20 to 80 years, with a majority being female. [Figure 1, image transcribed] 38,562 individuals aged≥18y from six NHANES cycles ... Missing values of VAI or sleep information(N=22,776) ... 15,786 individuals with VAI and sleep information ... Missing data of age, gender, race, marriage, education, poverty, smoke, alcohol use, physical activity, body mass index, healthy eating index, self-reported diseases (N=4,534) ... 11,252 individuals included in final analysis"
  },
  "exposure": {
   "definition": "Self-reported usual hours of sleep at night on weekdays or workdays, in three groups: short (< 7 h), middle (7-9 h, reference), long (> 9 h). Also continuous (per hour) in Supplementary Table S1.",
   "nhanes_variables": [
    "SLD010H (2007-2014)",
    "SLD012 (2015-2016, asked directly)",
    "SLD012 (2017-2018, derived from SLQ300/SLQ310)",
    "SLD012 (2021-2023, derived from SLQ300/SLQ310)"
   ],
   "nhanes_files": [
    "SLQ_E",
    "SLQ_F",
    "SLQ_G",
    "SLQ_H",
    "SLQ_I",
    "SLQ_J",
    "SLQ_L (2021-2023)"
   ],
   "transform": "three categories (middle reference)",
   "categories": "[Methods, Determination of sleep duration] short (less than 7 hours each day), middle (7–9 hours every day) and long (over 9 hours each day) [Results] Thus, we selected the middle sleep duration as our reference.",
   "quote": "[Methods, Determination of sleep duration] The amount of sleep was determined in the NHANES through the following question: ‘How much sleep do you usually get at night on weekdays or workdays?’. According to the American Academy of Sleep Medicine and the Sleep Research Society, self-reported sleep duration can be divided into three categories: short (less than 7 hours each day), middle (7–9 hours every day) and long (over 9 hours each day)."
  },
  "outcome": {
   "definition": "Visceral Adiposity Index, continuous and sex-specific: men VAI = (WC / (39.68 + 1.88 x BMI)) x (TG / 1.03) x (1.31 / HDL-C); women VAI = (WC / (36.58 + 1.89 x BMI)) x (TG / 0.81) x (1.52 / HDL-C); WC in cm, BMI in kg/m2, TG and HDL-C in mmol/L. The paper's formula has an extra closing parenthesis.",
   "nhanes_variables": [
    "BMXWAIST",
    "BMXBMI",
    "LBDTRSI (TRIGLY, fasting subsample; presumed, see unstated)",
    "LBDHDDSI",
    "RIAGENDR"
   ],
   "nhanes_files": [
    "BMX_E to BMX_J",
    "TRIGLY_E to TRIGLY_J",
    "HDL_E to HDL_J",
    "DEMO_E to DEMO_J",
    "BMX_L, TRIGLY_L, HDL_L, DEMO_L (2021-2023)"
   ],
   "quote": "[Methods, Assessment of VAI] The NHANES collected anthropometric data (height, weight and WC) and lipid parameters (TG, HDL-C) to calculate VAI according to previous studies. ... BMI was calculated by dividing a person’s weight in kilograms by the square of their height in metres. The formula for men was VAI=(WC/(39.68+1.88×BMI)))×(TG/1.03)×(1.31/HDL-C) and for women it was VAI=(WC/(36.58+1.89× BMI)))×(TG/0.81)×(1.52/HDL-C). The TG and HDL-C were expressed in mmol/L while the WC was expressed in cm."
  },
  "covariates": [
   {
    "name": "age",
    "coding": "years, continuous (Table 1 weighted means)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Model III (Model II adjusts for age only)"
   },
   {
    "name": "gender",
    "coding": "Female/Male",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "race",
    "coding": "white, Mexican, black, other",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "other = Other Hispanic plus other race/multiracial (presumed)"
   },
   {
    "name": "marital status",
    "coding": "unmarried/married",
    "nhanes_variables": [
     "DMDMARTL (2007-2018)",
     "DMDMARTZ (2021-2023)"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Where 'living with partner' falls is unstated; DMDMARTZ groups married with living with partner"
   },
   {
    "name": "education",
    "coding": "grade or less/high school/some college/college or more",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Mapping of the five DMDEDUC2 levels to four groups unstated ('grade or less' presumably < 9th grade plus 9-11th grade)"
   },
   {
    "name": "poverty",
    "coding": "family poverty income ratio, continuous (Table 1 weighted means)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "smoke status",
    "coding": "never/former/now",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "alcohol drinking status",
    "coding": "never/former/mild/moderate/heavy (definitions unstated)",
    "nhanes_variables": [
     "ALQ101, ALQ110, ALQ120Q/U, ALQ130, ALQ141Q/U (2007-2016)",
     "ALQ111, ALQ121, ALQ130, ALQ142, ALQ151 (2017-2018 and 2021-2023)"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "The questionnaire changed in 2017-2018; 2021-2023 matches 2017-2018"
   },
   {
    "name": "physical activity",
    "coding": "inactive/insufficient/sufficient (Table 1 reports min/week; cutpoints unstated)",
    "nhanes_variables": [
     "GPAQ items PAQ605-PAQ680 (unstated which)"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": false,
    "note": "2021-2023 PAQ_L has leisure-time moderate/vigorous items only (no work or transport activity)"
   },
   {
    "name": "energy intake",
    "coding": "kcal/day, continuous",
    "nhanes_variables": [
     "DR1TKCAL (day 1 or two-day mean unstated)"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "Both 2021-2023 recalls were by telephone"
   },
   {
    "name": "hypertension",
    "coding": "self-reported, yes/no",
    "nhanes_variables": [
     "BPQ020"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "diabetes",
    "coding": "self-reported, yes/no",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "Handling of 'borderline' unstated"
   },
   {
    "name": "stroke",
    "coding": "self-reported, yes/no",
    "nhanes_variables": [
     "MCQ160F"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "heart attack",
    "coding": "self-reported, yes/no",
    "nhanes_variables": [
     "MCQ160E"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "congestive heart failure",
    "coding": "self-reported, yes/no",
    "nhanes_variables": [
     "MCQ160B"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "coronary heart disease",
    "coding": "self-reported, yes/no",
    "nhanes_variables": [
     "MCQ160C"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "cancer",
    "coding": "self-reported, yes/no",
    "nhanes_variables": [
     "MCQ220"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "BMI",
    "coding": "underweight < 18.5, healthy 18.5-<25, overweight 25-<30, obese >= 30",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Listed as a covariate but not in Model III (BMI is part of VAI); used for subgroups"
   },
   {
    "name": "healthy diet (Healthy Eating Index-2015)",
    "coding": "HEI-2015 score",
    "nhanes_variables": [
     "computed from DR1IFF/DR1TOT with USDA food pattern equivalents"
    ],
    "nhanes_files": [
     "DR1IFF",
     "FPED (USDA, outside NHANES)"
    ],
    "in_2021_2023": false,
    "note": "Listed as a covariate but not in Model III; used only as a completeness requirement"
   }
  ],
  "design": {
   "weights": "WTMEC2YR x 1/6 (MEC exam weight, six cycles)",
   "strata_psu": "complex design accounted for ('weighted analysis ... to address the intricate sampling design'); SDMVSTRA/SDMVPSU not named",
   "quote": "[Methods, Statistical analysis] In compliance with NHANES recommendations, a weighted analysis was performed to address the intricate sampling design in the study. The sampling weight was calculated by multiplying the 2-year weights of the Mobile Exam Centre by one-sixth. ... Statistical analyses were conducted using R statistical software (V.4.2.2, http://www.R-project.org) and EmpowerStats (http://www.empowerstats.com). All results were deemed significant at p<0.05.",
   "software": "R 4.2.2 and EmpowerStats",
   "missing_data": "complete case",
   "quote_missing": "[Methods, Study population] Participants who had missing data for VAI, sleep duration or other important covariates at the baseline assessment were excluded."
  },
  "model": {
   "family": "linear (Gaussian generalized linear model)",
   "weighted": true,
   "quote": "[Methods, Statistical analysis] The weighted multiple generalised linear regression was used to assess the association between sleep duration (both as a categorical and continuous variable) and VAI. The results were expressed as β and 95% CIs."
  },
  "unstated": [
   "Whether triglycerides are the fasting-subsample values (TRIGLY files; the 15,786 of 38,562 adults with VAI suggests so) and, if so, why the MEC weight rather than the fasting-subsample weight (WTSAF2YR) was used",
   "Strata and PSU variables",
   "How the category boundaries treat half-hour values from 2015-2018 (e.g. 6.5 h, 9.5 h) and exactly 9 h (presumably middle)",
   "Which recall day(s) give energy intake",
   "Definitions of physical activity groups and alcohol drinking groups",
   "Education and marital status mappings (where 'living with partner' falls)",
   "How age, poverty income ratio and energy intake entered the model (presumably linear continuous)",
   "Whether 'borderline' diabetes counts, and whether pregnant women were excluded",
   "Whether 18-19-year-olds were eligible (stated >= 18; analyzed range 20-80)",
   "VAI handling of extreme values (none described)"
  ],
  "notes": "Table A's predictor 'Sleep health' is in fact self-reported weekday/workday sleep duration only; 2021-2023 SLD012 is derived from usual sleep and wake times (SLQ300/SLQ310) like 2017-2018, while 2007-2016 asked the hours directly, so the instrument changed but not the construct or unit. Headline rule: sleep duration has a middle reference, so the first contrast in the abstract (short vs middle) stands. The continuous estimate (-0.05 per hour, 95% CI -0.09 to -0.01, Model III) is only in Supplementary Table S1 (scratchpad/dl/supp/row133/files/bmjopen-14-7-s001.pdf), so it is not a headline candidate. Figure 1 (image) gives the flow; figures and supplement were fetched from the PMC open-access S3 bucket (pmc-oa-opendata) into scratchpad/dl/supp/row133/. The GAM inflection at 7.5 h/day has no segment estimates or CIs in the text. Typos: Table 2 Model II long-sleep CI '0.20 (0.09 to 0.49); 0.17' (the lower bound must be negative); Table 1 alcohol rows 'Never' and 'Heavy' are identical (15.27, 11.09, 15.17); the VAI formula has an unmatched ')'. Software includes EmpowerStats. 2021-2023 gaps: physical activity (Model III covariate) as measured by the GPAQ, and HEI-2015 (completeness requirement only). The paper uses 2007-2008 to 2017-2018 only; no 2017-March 2020 pre-pandemic (P_) files, so no overlap issue.",
  "adjudication": null
 },
 {
  "id": "row220",
  "rank": 130,
  "row": 220,
  "doi": "10.1007/s11255-023-03917-2",
  "pmcid": "PMC11090932",
  "title": "Association between a body shape index and prostate cancer: a cross-sectional study of NHANES 2001–2018",
  "authors": [
   "Liu, Xiaowu",
   "Shi, Honglei",
   "Shi, Yunfeng",
   "Wei, Hanping",
   "Yuan, Xiaoliang",
   "Jiao, Zhimin",
   "Wu, Tingchun",
   "Wang, Zengjun"
  ],
  "year": 2024,
  "journal": "International Urology and Nephrology",
  "table_a": {
   "predictor": "Body shape index",
   "condition": "Prostate cancer",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "When comparing the second, third, and fourth ABSI quartile to the lowest quartile, the adjusted odds ratios (95% confidence intervals) for PCa risk were 1.34 (0.77, 2.31), 1.75 (1.03, 3.00), and 1.91 (1.12, 3.27), respectively (p for trend = 0.011).",
   "table_location": "Table 2 (Association between ABSI and PCa in NHANES 2001–2018), row 'Q4 (86.42, 116.59)', column 'Model IIIc' (OR (95% CI) and p value)",
   "table_quote": "| Q4 (86.42, 116.59) | 6.43 (3.96, 10.42) | < 0.001 | 1.92 (1.15, 3.23) | 0.014 | 1.91 (1.12, 3.27) | 0.019 |",
   "measure": "OR",
   "estimate": 1.91,
   "ci_low": 1.12,
   "ci_high": 3.27,
   "p_value": "0.019",
   "exposure_contrast": "highest vs lowest ABSI quartile: Q4 (86.42-116.59) vs Q1 (66.74-80.94, reference), ABSI scaled by 1000; the abstract lists Q2, Q3 and Q4 vs Q1, and the quartile coding is represented by its highest category",
   "model_label": "Model III",
   "covariates_in_this_model": [
    "age",
    "race",
    "education level",
    "family income level",
    "living status",
    "BMI",
    "drinking status",
    "smoking status",
    "hypertension",
    "diabetes"
   ],
   "n_analytic": 11013,
   "n_quote": "[Abstract, Methods] 11,013 participants were enrolled in the National Health and Nutrition Examination Survey from 2001 to 2018. [Results] A total of 11,013 participants were enrolled in this study, 492 with PCa and 10,521 without PCa. [Fig. 1, image transcribed] Final subjects included (n=11013)",
   "events": 492
  },
  "cycles": [
   "2001-2002",
   "2003-2004",
   "2005-2006",
   "2007-2008",
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=40",
   "defining": "men aged 40 or older (with no history of a cancer other than prostate cancer)",
   "inclusion": "Male NHANES 2001-2018 participants aged 40 or older with no other tumor history except prostate cancer and complete covariate data",
   "exclusions": [
    "Female participants (n=46341): 91351 -> 45010",
    "Age < 40 years (n=28883) -> 16127",
    "History of other tumors or missing data on covariates (n=5114, not split) -> 11013"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "[Methods, Study design and population] The inclusion criteria for all individuals were men aged ≥ 40 years old, with no other tumor history except PCa. Ultimately, after excluding participants with missing data for important covariates, a total of 11,013 participants were enrolled (Fig. 1). [Fig. 1, image transcribed] NHANES 2001-2018 (n=91351) ... Excluded: Female participants (n=46341) ... NHANES 2001-2018 45010 subjects included ... Excluded: Age< 40 years old (n=28883) ... NHANES 2001-2018 16127 subjects included ... Excluded: With history of other tumors or miss data on covariates (n=5114) ... Final subjects included (n=11013)"
  },
  "exposure": {
   "definition": "A body shape index, ABSI = 1000 x WC (m) x height (m)^(5/6) x weight (kg)^(-2/3), i.e. 1000 x WC / (BMI^(2/3) x height^(1/2)) with WC and height in metres (Table 1 mean 83.40). Analyzed continuously (per unit of the x1000 scale) and in quartiles; the headline uses quartiles.",
   "nhanes_variables": [
    "BMXWAIST",
    "BMXHT",
    "BMXWT"
   ],
   "nhanes_files": [
    "BMX_B to BMX_J",
    "BMX_L (2021-2023)"
   ],
   "transform": "ABSI x 1000; quartiles (Q1 reference); also continuous",
   "categories": "[Table 2, row labels] Q1 (66.74, 80.94) ... Q2 (80.94, 83.66) ... Q3 (83.66, 86.42) ... Q4 (86.42, 116.59) [Methods, Statistical analysis] The same approach was employed to explore the relationship in the ABSI quartiles, with the lowest quartile as the reference.",
   "quote": "[Methods, Variables] The dependent variable was PCa, and the independent variable in this research was ABSI. ... ABSI was calculated by ... $$1000\\times{\\text{WC}}\\;{\\text{(m)}} \\times {\\text{height}}\\;{\\text{(m)}}^\\frac{5}{6} \\times {\\text{weight}}\\;{\\text{(kg}})^{ - \\frac{2}{3}}$$"
  },
  "outcome": {
   "definition": "Self-reported ever doctor-diagnosed prostate cancer: MCQ220 = 1 and a cancer type of prostate (code 30) in MCQ230A/B/C; men with any other cancer were excluded, so controls are men with no cancer history",
   "nhanes_variables": [
    "MCQ220",
    "MCQ230A",
    "MCQ230B",
    "MCQ230C",
    "MCQ230D"
   ],
   "nhanes_files": [
    "MCQ_B to MCQ_J",
    "MCQ_L (2021-2023)"
   ],
   "quote": "[Methods, Variables] The questionnaire items, “Ever been told you had cancer or a malignancy of any kind” and “What kind of cancer was it” were used to identify individuals with PCa."
  },
  "covariates": [
   {
    "name": "age",
    "coding": "< 60, 60-80, >= 80 years (Table 1); model coding unstated",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Top-coded at 85 in 2001-2006 and at 80 from 2007 (and in 2021-2023)"
   },
   {
    "name": "race",
    "coding": "Non-Hispanic white, Non-Hispanic black, Mexican American, Other race, Other Hispanic",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Reference category unstated"
   },
   {
    "name": "education level",
    "coding": "less than high school vs high school or above (Methods); <= high school vs > high school (Table 1)",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Methods and Table 1 disagree on where high school graduates go"
   },
   {
    "name": "family income level",
    "coding": "low (PIR <= 1), middle (1 < PIR < 3), high (PIR >= 3)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "living status",
    "coding": "living alone (widowed, divorced, separated, never married) vs living with a partner (married, living with a partner)",
    "nhanes_variables": [
     "DMDMARTL (2001-2018)",
     "DMDMARTZ (2021-2023)"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DMDMARTZ groups exactly into these two sets"
   },
   {
    "name": "BMI",
    "coding": "under/normal weight (< 25), overweight (25 to < 30), obese (>= 30)",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "drinking status",
    "coding": "heavy (>= 3 drinks/day, or binge >= 5 drinks on 5 or more days per month), never (< 12 drinks in lifetime), mild (others)",
    "nhanes_variables": [
     "ALQ101/ALQ110 (2001-2016)",
     "ALQ111 (2017-2018)",
     "ALQ130",
     "ALQ141Q/U (2001-2016)",
     "ALQ142 (2017-2018)"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "2021-2023 ALQ_L has ALQ111 (ever 1 drink) but no 12-drinks-in-lifetime item, so 'never' needs ALQ111; ALQ130 and ALQ142/ALQ170 give heavy and binge"
   },
   {
    "name": "smoking status",
    "coding": "never (< 100 cigarettes), former (>= 100, not now), current (>= 100, smoking now)",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "hypertension",
    "coding": "self-reported hypertension, antihypertensive drugs, SBP >= 140 mmHg, or DBP >= 80 mmHg (sic)",
    "nhanes_variables": [
     "BPQ020",
     "BPQ040A/BPQ050A (2001-2018)",
     "BPQ150 (2021-2023)",
     "BPXSY1-4, BPXDI1-4 (2001-2018)",
     "BPXOSY1-3, BPXODI1-3 (2021-2023)"
    ],
    "nhanes_files": [
     "BPQ",
     "BPX",
     "BPXO_L"
    ],
    "in_2021_2023": true,
    "note": "2021-2023 blood pressure is oscillometric; which readings were averaged is unstated"
   },
   {
    "name": "diabetes",
    "coding": "self-reported diabetes, HbA1c >= 6.5%, fasting glucose >= 7.0 mmol/l, 2-h OGTT or random glucose >= 11.1 mmol/l, or antidiabetic medication",
    "nhanes_variables": [
     "DIQ010",
     "LBXGH",
     "LBXGLU",
     "LBXGLT (OGTT, 2005-2016)",
     "DIQ050",
     "DIQ070"
    ],
    "nhanes_files": [
     "DIQ",
     "GHB",
     "GLU",
     "OGTT"
    ],
    "in_2021_2023": true,
    "note": "No OGTT in 2021-2023 (refrigerated serum glucose LBXSGL could serve as the random glucose); other components present"
   }
  ],
  "design": {
   "weights": "unstated: 'taking sample weights into consideration, according to the CDC guidelines' (which weight, the 4-year weight for 2001-2002, and the divisor for nine cycles are not given)",
   "strata_psu": "unstated (design variables not named)",
   "quote": "[Methods, Statistical analysis] Statistical analysis was conducted by taking sample weights into consideration, according to the Centers for Disease Control and Prevention (CDC) guidelines. ... R 4.3.1 (https://www.r-project.org/) was used to carry out all of the data analysis. A two-sided p < 0.05 was considered statistically significant.",
   "software": "R 4.3.1",
   "missing_data": "complete case",
   "quote_missing": "[Methods, Study design and population] Ultimately, after excluding participants with missing data for important covariates, a total of 11,013 participants were enrolled (Fig. 1)."
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "[Methods, Statistical analysis] Secondly, weighted multivariate logistic regression was utilized to investigate the association between ABSI and PCa, and three statistical models were constructed: model I, adjusted with no covariates; model II, just adjusted for demographic covariates only; and model III, adjusted for all covariates. The same approach was employed to explore the relationship in the ABSI quartiles, with the lowest quartile as the reference. The final results are represented as odds ratios (OR) and 95% confidence intervals (CIs)."
  },
  "unstated": [
   "Which sample weight (MEC or interview; WTMEC4YR or WTMEC2YR for 2001-2002) and how the nine cycles' weights were divided",
   "Strata and PSU variables",
   "Whether ABSI quartile cutpoints were weighted or unweighted, and which quartile a boundary value falls in (printed ranges share endpoints)",
   "How age entered the models (three groups as in Table 1, or continuous)",
   "Reference categories for race and the other categorical covariates",
   "How men with prostate cancer plus another cancer, and cancer type answered 'don't know' or refused, were handled",
   "NHANES items behind 'never drinker' in 2017-2018 (no lifetime 12-drink item) and the binge definition",
   "Which glucose measures (fasting subsample, OGTT, 'random blood glucose') and medication items define diabetes, and which blood pressure readings define hypertension",
   "Exact Model III sample per quartile (only the total 11,013 and 492 cases are given)"
  ],
  "notes": "Headline rule: the abstract's first coding is ABSI quartiles with the lowest quartile as reference, so it is represented by Q4 vs Q1 (Model III OR 1.91, 1.12-3.27, p = 0.019); the abstract lists Q2 (1.34, 0.77-2.31, not significant) and Q3 (1.75, 1.03-3.00) first. Continuous ABSI (per 1 unit of ABSI x 1000) Model III OR 1.05 (1.02-1.08), p = 0.003. The formula renders badly in the text ('1000×WC(m)×height(m)56×weight(kg)-23'); the LaTeX source gives height^(5/6) and weight^(-2/3). Typos: the Results heading reads 'Associations between BRI and PCa'; Table 1 lists 'Weight (cm)'; the hypertension DBP cutoff is printed >= 80 mmHg; education is '< high school vs >= high school' in the Methods but '<= vs > high school level' in Table 1. 2021-2023 notes: blood pressure is oscillometric (BPXO_L), antihypertensive use is BPQ150, no OGTT, and ALQ_L lacks the lifetime-12-drinks item; none of these touch the exposure, outcome, or population. Figures and supplements (Supplementary Tables 1-2, Figure 1) were fetched from Springer static content and the PMC open-access S3 bucket into scratchpad/dl/supp/row220/. The paper uses 2001-2002 to 2017-2018 only; no 2017-March 2020 pre-pandemic (P_) files, so no overlap issue.",
  "adjudication": null
 },
 {
  "id": "row275",
  "rank": 136,
  "row": 275,
  "doi": "10.1097/MD.0000000000030539",
  "pmcid": "PMC9509149",
  "title": "Association of medical uninsurance with sociodemographic attributes in US cancer population: A cross-sectional study of NHANES data 2013 to 2018",
  "authors": [
   "Wahab, Ahsan",
   "Abdelazeem, Basel",
   "Masood, Adeel",
   "Khakwani, Maria",
   "Kumar Jakka, Bharath",
   "Koduru, Ujwala",
   "Ehsan, Hamid"
  ],
  "year": 2022,
  "journal": "Medicine",
  "table_a": {
   "predictor": "Medical uninsurance",
   "condition": "Sociodemographic attributes",
   "population": "US adults with cancer"
  },
  "headline": {
   "abstract_quote": "In the multivariate model, age (AOR: 0.95, 95% CI: 0.93–0.96), female sex (AOR: 2.88, 95% CI: 1.25–6.62), <high school education (AOR: 4.02, 95% CI: 1.24–13.00), and non-US-born status with <20-years stay (AOR: 3.42, 95% CI: 1.44–8.11) were independent predictors of MU.",
   "table_location": "Table 3 (Multivariate logistic regression analysis for the predictors of medical uninsurance in the US cancer population from NHANES data 2013–2018), row 'Age', column 'AOR (95% CI) of uninsurance vs insurance'",
   "table_quote": "| Age | 0.95 (0.93–0.96) |",
   "measure": "OR",
   "estimate": 0.95,
   "ci_low": 0.93,
   "ci_high": 0.96,
   "p_value": "not reported",
   "exposure_contrast": "per 1-year increase in age (continuous), odds of being uninsured vs insured among adults with a cancer history; the per-year unit is stated for the unadjusted estimate ('6% lower in 1-year older participants') and presumed for the adjusted one",
   "model_label": "multivariate logistic regression model (Table 3, the paper's only adjusted model)",
   "covariates_in_this_model": [
    "Gender",
    "Race/Ethnicity",
    "Education",
    "US birth/length status",
    "Annual household income"
   ],
   "n_analytic": 1681,
   "n_quote": "[Results] The total sampled US cancer population was 1681 participants (weighted US population: 25,982,352), with 4.3% ± 0.62 being uninsured (weighted US population of 1,129,421). [Table 1 header] | n* = 1681 <colspan=3> | n = 1596 <colspan=3> | n = 85 <colspan=3> | [Table 1 note] Missing values: Education = 3, Annual Household Income = 104, US Birth/Length Status = 10. [comment: study population; the multivariable model's N is not reported and is smaller, since cases missing education, income or US birth/length status drop out]",
   "events": 85
  },
  "cycles": [
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=20",
   "defining": "US adults (aged 20 or older) with a history of cancer (ever told they had cancer or a malignancy of any kind)",
   "inclusion": "NHANES 2013-2018 participants aged 20 or older who reported any previous or current cancer (MCQ220 = 1) and disclosed their insurance status",
   "exclusions": [
    "Not ever told they had cancer (29,400 -> 1,684 ever told they had cancer)",
    "Without insurance information (N = 3) -> 1,681 (1,596 insured, 85 uninsured)",
    "In the multivariable model, cases missing education (3), annual household income (104) or US birth/length status (10) drop out (NOMCAR keeps them only for variance estimation)"
   ],
   "exclusions_not_in_2021_2023": [
    "The model's complete-case exclusion on annual household income: INDHHIN2 is not in DEMO_L 2021-2023"
   ],
   "quote": "[Materials and Methods, 2.2 Inclusion/exclusion criteria and sociodemographic attributes] Those participants (aged 20 years and above) who reported any previous or current history of cancer (ever told you had cancer or malignancy of any kind) met the inclusion criteria. The outcome of interest was medical uninsurance (are you covered by health insurance or some other kind of health care plan, irrespective of the type of coverage). We excluded participants who did not disclose medical insurance information (yes vs no). [Figure 1, image transcribed] Total NHANES participants 2013-2018 N = 29,400 ... Participants ever told they had cancer N = 1,684 ... Participants without insurance information N = 3 ... Participants with insurance information N = 1,681 ... Participants with insurance N = 1,596 ... Participants without insurance N = 85"
  },
  "exposure": {
   "definition": "Age in years at screening, continuous. Chosen by the headline rule as the first of the paper's six sociodemographic predictors (age, sex, race/ethnicity, education, annual household income, US birth/length of stay), which enter the model together.",
   "nhanes_variables": [
    "RIDAGEYR"
   ],
   "nhanes_files": [
    "DEMO_H",
    "DEMO_I",
    "DEMO_J",
    "DEMO_L (2021-2023)"
   ],
   "transform": "none (continuous, per year)",
   "categories": null,
   "quote": "[Materials and Methods, 2.2] We evaluated the association of sociodemographic correlates, such as age, sex, education, race/ethnicity, annual household income, and US birth/length status, with MU. [Results] Using bivariate logistic regression models (Table 2), we determined the association between sociodemographic factors as predictors and MU as outcome and found that the odds of being uninsured were 6% lower in 1-year older participants (UOR: 0.94, 95% CI: 0.93–0.96)."
  },
  "outcome": {
   "definition": "Medically uninsured: answered no to being covered by health insurance or any other kind of health care plan (HIQ011 = 2) vs covered (HIQ011 = 1), irrespective of coverage type",
   "nhanes_variables": [
    "HIQ011"
   ],
   "nhanes_files": [
    "HIQ_H",
    "HIQ_I",
    "HIQ_J",
    "HIQ_L (2021-2023)"
   ],
   "quote": "[Materials and Methods, 2.2] The outcome of interest was medical uninsurance (are you covered by health insurance or some other kind of health care plan, irrespective of the type of coverage). We excluded participants who did not disclose medical insurance information (yes vs no). [Materials and Methods, 2.3 Statistical analyses] We divided the cancer population with insurance information into 2 groups: medically insured and uninsured."
  },
  "covariates": [
   {
    "name": "Gender",
    "coding": "Male (reference), Female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "Race/Ethnicity",
    "coding": "Non-Hispanic White (reference), Non-Hispanic Black, Non-Hispanic other races (Non-Hispanic Asian plus other race including multiracial), Hispanics (Mexican American plus other Hispanic)",
    "nhanes_variables": [
     "RIDRETH3"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "RIDRETH3 is in DEMO_L with the same codes"
   },
   {
    "name": "Education",
    "coding": "College graduates (reference), <High school, High school/GED, Some college/associate degree",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "<High school presumably DMDEDUC2 1-2"
   },
   {
    "name": "US birth/length status",
    "coding": "US born (reference); non-US born with length of stay < 20, 20-39, and >= 40 years",
    "nhanes_variables": [
     "DMDBORN4",
     "DMDYRSUS (2013-2018)",
     "DMDYRUSR (2021-2023)"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": false,
    "note": "2021-2023 DMDYRUSR stops at '20 years or more', so US born / < 20 / >= 20 can be built but not the 20-39 vs >= 40 split"
   },
   {
    "name": "Annual household income",
    "coding": "<$20K (reference), $20K-$44,999, $45K-$74,999, $75K-$99,999, >= $100K",
    "nhanes_variables": [
     "INDHHIN2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": false,
    "note": "DEMO_L has no household income in dollar ranges (only INDFMPIR; INQ_L has the monthly poverty index); INDHHIN2 codes 12 ('$20,000 and over') and 13 ('Under $20,000') make the grouping partly ambiguous even in 2013-2018"
   }
  ],
  "design": {
   "weights": "WTINT2YR/3 (interview weight, three cycles)",
   "strata_psu": "SDMVSTRA (strata) and SDMVPSU (cluster)",
   "quote": "[Materials and Methods, 2.1] To incorporate the fact that data spanned over 6 years and 3 cycles, we divided the total weighted sample by 3. [Materials and Methods, 2.3] Proc survey procedures were used to perform data analysis to incorporate the survey design features. We used interview weights (WTINT2YR/3), strata (SDMVSTRA), and cluster (SDMVPSU) variables during survey procedures. For subgroup or subset analyses, we used table statement in the surveyfreq procedure, whereas domain statement in the survey means procedure. Missing values were dealt with in the NOMCAR statement, as recommended by the NHANES. Data analysis was performed using SAS/STAT software, version 9.4, Copyright (c) 2016 by SAS Institute Inc., Cary, NC, USA.",
   "software": "SAS/STAT 9.4 (survey procedures)",
   "missing_data": "complete case in the model, with the NOMCAR option (missing-value cases kept only for variance estimation)",
   "quote_missing": "[Materials and Methods, 2.3] Missing values were dealt with in the NOMCAR statement, as recommended by the NHANES. [Table 1 note] Missing values: Education = 3, Annual Household Income = 104, US Birth/Length Status = 10."
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "[Materials and Methods, 2.3] We performed a bivariate logistic regression model to evaluate the association of individual sociodemographic attributes and MU and calculated the unadjusted odds ratios (UOR) of uninsurance. Finally, we performed a multivariate logistic regression model, evaluated the association of the same sociodemographic attributes (when considered together) and MU, and calculated the adjusted odds ratio (AOR)."
  },
  "unstated": [
   "The multivariable model's sample size and number uninsured (complete cases on education, income and US birth/length status)",
   "Whether the cancer subpopulation was analyzed with a DOMAIN statement on the full sample or by subsetting the data (this changes the variance)",
   "The unit and form of age in the adjusted model (presumably continuous, per year)",
   "How INDHHIN2 codes 12 ('$20,000 and over'), 13 ('Under $20,000'), refused and don't know were grouped",
   "How refused/don't know answers to country of birth, length of stay and insurance were treated (the 3 without insurance information presumably include them)",
   "The p-value of the age AOR"
  ],
  "notes": "Predictor and condition worked out: the paper models medical uninsurance (HIQ011, not covered by any health insurance or plan) as the outcome, with six sociodemographic attributes as mutually adjusted predictors (age, sex, race/ethnicity, education, annual household income, US birth/length of stay), in adults aged 20+ with a history of cancer (MCQ220), the condition that names the population. Suchak et al.'s labels reverse the roles and name a set of variables as the condition; neither label is a health outcome. By the headline rule the first coding in the abstract's Results is age (continuous), giving AOR 0.95 (0.93-0.96); the other significant adjusted predictors are female sex 2.88 (1.25-6.62), < high school vs college graduate 4.02 (1.24-13.00), and non-US born with < 20 years in the US vs US born 3.42 (1.44-8.11). For age, the other model terms are covariates, two of which (annual household income; the 20-39 vs >= 40 years-in-US split) cannot be built in 2021-2023. The abstract is missing from the text file (its abstract section holds only keywords) and was taken from the Europe PMC XML (scratchpad/dl/fulltext/row275_PMC9509149.xml), matching the published PDF. Figures were fetched from the PMC open-access S3 bucket into scratchpad/dl/supp/row275/. Small errors in Table 1 and Results ('27.6 ± −2.1', '9.7% ± −3.1', '92.0 +/1.0', footnote '*acute individual'). The paper uses 2013-2014 to 2017-2018 and says it left out 2019-2020; no 2017-March 2020 pre-pandemic (P_) files, so no overlap issue.",
  "adjudication": null
 },
 {
  "id": "row284",
  "rank": 147,
  "row": 284,
  "doi": "10.1186/s12889-022-13524-y",
  "pmcid": "PMC9161202",
  "title": "Association between sleep duration and albumin in US adults: a cross-sectional study of NHANES 2015–2018",
  "authors": [
   "Li, Jingxian",
   "Guo, Lizhong"
  ],
  "year": 2022,
  "journal": "BMC Public Health",
  "table_a": {
   "predictor": "Sleep health",
   "condition": "Albumin",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "Compared to 7–8 h of sleep, short sleep duration was linked to lower albumin levels [sleep duration ≤ 5 h: β =-1.00, 95% CI (-1.26, -0.74), P < 0.0001].",
   "table_location": "Table 2 ('Relationship between sleep duration/ sleep onset timing and albumin in different models'), section 'sleep duration (h)', row '≤ 5' (reference row '7–8'), column 'Model 3' (β(95%CI) and P value).",
   "table_quote": "| ≤ 5 | -1.09 (-1.43, -0.76) | < 0.0001 | -0.93 (-1.24, -0.61) | < 0.0001 | -1.00 (-1.26, -0.74) | < 0.0001 |",
   "measure": "beta",
   "estimate": -1.0,
   "ci_low": -1.26,
   "ci_high": -0.74,
   "p_value": "< 0.0001",
   "exposure_contrast": "weekday/workday sleep duration <= 5 h vs 7-8 h (7 h < sleep duration <= 8 h, reference), from a model with all six categories (<= 5, 5-6, 6-7, 7-8, 8-9, > 9 h); beta is the difference in serum albumin in g/L",
   "model_label": "Model 3",
   "covariates_in_this_model": [
    "sex",
    "age",
    "race",
    "marital status",
    "moderate work activity",
    "TP",
    "ALT",
    "AST",
    "Cr",
    "UACR",
    "HS-CRP",
    "GLU",
    "BMI",
    "hypertension",
    "high cholesterol",
    "cancer or malignancy"
   ],
   "n_analytic": 9973,
   "n_quote": "[Abstract] A total of 9,973 participants aged ≥ 20 years were included in this study from NHANES 2015–2018. [Study design] Through the rigorous screening, of 19,225 participants, a total of 9,973 participants aged ≥ 20 years with a complete set of sleep-related data and albumin data were included in this study. [Fig. 1 (image; transcribed in dl/supp/row284/transcribed_from_images.txt)] Enrolled analyses (n=9,973) [comment: Table 2 does not print the number of participants in each model; whether Model 3 lost participants with missing covariates is not stated.]",
   "events": null
  },
  "cycles": [
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=20",
   "defining": "US adults aged 20 years or older; no condition or characteristic defines the population",
   "inclusion": "NHANES 2015-2018 participants aged >= 20 with complete weekday sleep duration, weekday sleep timing and serum albumin data",
   "exclusions": [
    "age < 20 y: 7,937 (19,225 to 11,288, Fig. 1)",
    "incomplete data of sleep duration: 79 and incomplete data of sleep timing: 35 (114 in all; 11,288 to 11,174, Fig. 1)",
    "missing albumin data: 1,201 (11,174 to 9,973, Fig. 1)"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "[Abstract] A total of 9,973 participants aged ≥ 20 years were included in this study from NHANES 2015–2018. [Study design] Through the rigorous screening, of 19,225 participants, a total of 9,973 participants aged ≥ 20 years with a complete set of sleep-related data and albumin data were included in this study. The detailed screening process of participants is shown in Fig. 1.\n[Fig. 1 (image), transcribed box text, one box per line:]\nParticipants of NHANES from 2015-2018 (n=19, 225)\nExcluded age < 20 y (n=7,937)\nAge ≥ 20 y (n=11,288)\nExcluded (n=114)\nIncomplete data of sleep duration (n=79)\nIncomplete data of sleep timing (n=35)\nComplete data of sleep duration and sleep timing (n=11,174)\nExcluded for missing albumin data (n=1,201)\nEnrolled analyses (n=9,973)"
  },
  "exposure": {
   "definition": "Self-reported usual sleep duration on weekdays or workdays, in hours (half-hour steps), NHANES SLD012 (the paper writes 'SLQ012'). Categorized into six groups: <= 5 h, 5 < h <= 6, 6 < h <= 7, 7 < h <= 8 (reference), 8 < h <= 9, > 9 h. In 2015-2016 (SLQ_I) SLD012 is the direct answer to 'How much sleep {do you/does SP} usually get at night on weekdays or workdays?' (2 to 14.5); in 2017-2018 (SLQ_J) it is derived from usual sleep and wake times SLQ300/SLQ310 (3 to 13.5; 2 = less than 3 hours; 14 = 14 hours or more).",
   "nhanes_variables": [
    "SLD012"
   ],
   "nhanes_files": [
    "SLQ (SLQ_I, SLQ_J; SLQ_L in 2021-2023)"
   ],
   "transform": "categories (six groups, 7-8 h reference); the paper also fits sleep duration as continuous hours and a two-segment model with a knot at 7.5 h (Table 3)",
   "categories": "[Sleep duration and sleep timing] The sleep duration was classified and presented in 6 groups: sleep duration ≤ 5 h, 5 h < sleep duration ≤ 6 h, 6 h < sleep duration ≤ 7 h, 7 h < sleep duration ≤ 8 h, 8 h < sleep duration ≤ 9 h, sleep duration > 9 h. They are described as follows: ≤ 5 h, 5–6 h, 6–7 h, 7–8 h, 8-9 h, > 9 h [21]. (“short” sleep duration: ≤ 5 h; “long” sleep duration: > 9 h) [22, 23].",
   "quote": "[Sleep duration and sleep timing] The definition was based on the “sleep disorders” data in the “NHANES Questionnaire”. Question SLQ012 asked “Number of hours usually sleep on weekdays or workdays.” The participants answered with their sleep duration, and hours were rounded to the nearest half-hour. Sleep duration was analysed as both a continuous and categorical variable."
  },
  "outcome": {
   "definition": "Serum albumin (ALB) in g/L from the standard biochemistry profile (bromocresol purple dye method), continuous.",
   "nhanes_variables": [
    "LBDSALSI (g/L; LBXSAL is the same in g/dL)"
   ],
   "nhanes_files": [
    "BIOPRO (BIOPRO_I, BIOPRO_J; BIOPRO_L in 2021-2023)"
   ],
   "quote": "[Albumin] Albumin is abbreviated as ALB. The method for measuring albumin concentration used the dye bromocresol purple. The NHANES Quality Control and Quality Assurance Protocol (QA/QC) complies with the requirements of the Clinical Laboratory Improvement Act of 1988. [Table 2 note] β was the effect size (g/L) of the change in albumin"
  },
  "covariates": [
   {
    "name": "sex",
    "coding": "male, female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Models 2 and 3."
   },
   {
    "name": "age",
    "coding": "unstated (Table 1 reports mean ± SE in years; presumably continuous)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Models 2 and 3."
   },
   {
    "name": "race",
    "coding": "Mexican American, non-Hispanic white, non-Hispanic black, other race",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Models 2 and 3. Variable not named; four groups, so where Other Hispanic went (RIDRETH1 = 2) is unstated, presumably 'other race'."
   },
   {
    "name": "marital status",
    "coding": "married or living with a partner; living alone (never married, separated, divorced, or widowed)",
    "nhanes_variables": [
     "DMDMARTL (2015-2018)",
     "DMDMARTZ (2021-2023)"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Models 2 and 3. DMDMARTZ (married/living with partner; widowed/divorced/separated; never married) yields the same two groups."
   },
   {
    "name": "moderate work activity",
    "coding": "yes or no, from 'does your work involve moderate-intensity activity that causes small increases in breathing or heart rate?'",
    "nhanes_variables": [
     "PAQ620"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": false,
    "note": "Models 2 and 3. PAQ_L asks only about leisure-time activity (PAD790-PAD820) and sedentary time; there are no work-activity questions in 2021-2023."
   },
   {
    "name": "TP (total protein)",
    "coding": "continuous, g/L (Table 1)",
    "nhanes_variables": [
     "LBDSTPSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Model 3. Cobas 8000 bridging: no adjustment recommended."
   },
   {
    "name": "ALT",
    "coding": "continuous, IU/L (Table 1)",
    "nhanes_variables": [
     "LBXSATSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Model 3. The CDC recommends a Deming equation to bridge Cobas 6000 and Cobas 8000 ALT."
   },
   {
    "name": "AST",
    "coding": "continuous, IU/L (Table 1)",
    "nhanes_variables": [
     "LBXSASSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "Cr (serum creatinine)",
    "coding": "μmol/L; Table 1 shows log2-transformed values ('Crlog2'); the model's coding is unstated",
    "nhanes_variables": [
     "LBDSCRSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "UACR",
    "coding": "mg/g; Table 1 shows log2-transformed values ('UACRlog2'); the model's coding is unstated",
    "nhanes_variables": [
     "URDACT (or URXUMA/URXUCR)"
    ],
    "nhanes_files": [
     "ALB_CR"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "HS-CRP",
    "coding": "continuous, mg/L (Table 1)",
    "nhanes_variables": [
     "LBXHSCRP"
    ],
    "nhanes_files": [
     "HSCRP"
    ],
    "in_2021_2023": true,
    "note": "Model 3. HSCRP_I/HSCRP_J in 2015-2018, HSCRP_L in 2021-2023."
   },
   {
    "name": "GLU (glucose)",
    "coding": "continuous, mmol/L (Table 1)",
    "nhanes_variables": [
     "LBDSGLSI (serum, biochemistry) or LBDGLUSI (fasting plasma, GLU file): unstated which"
    ],
    "nhanes_files": [
     "BIOPRO or GLU"
    ],
    "in_2021_2023": true,
    "note": "Model 3. Both versions exist in 2021-2023 (fasting glucose only in the fasting subsample)."
   },
   {
    "name": "BMI",
    "coding": "continuous, kg/m2, weight in kilograms divided by height in meters squared",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "hypertension",
    "coding": "self-reported doctor or health professional diagnosis: yes, no, unknown",
    "nhanes_variables": [
     "BPQ020"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3. Table 1 keeps an 'unknown' category."
   },
   {
    "name": "high cholesterol",
    "coding": "self-reported doctor or health professional diagnosis: yes, no, unknown",
    "nhanes_variables": [
     "BPQ080"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3. Table 1 keeps an 'unknown' category."
   },
   {
    "name": "cancer or malignancy",
    "coding": "self-reported doctor or health professional diagnosis: yes, no, unknown",
    "nhanes_variables": [
     "MCQ220"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3. Table 1 keeps an 'unknown' category."
   }
  ],
  "design": {
   "weights": "WTMEC2YR divided by 2 (MEC4YR = 1/2 x WTMEC2YR) for the combined 2015-2018 MEC sample",
   "strata_psu": "unstated (only the weights are described; SDMVSTRA and SDMVPSU are not mentioned)",
   "quote": "[Statistical analysis] Weighted data were calculated according to analytical guidelines (NHANES:2015–2016 and 2017–2018). Various sample weights, such as interview weight (wtint2yr), MEC (Mobile Examination Center) exam weight (wtmec2yr), and several subsample weights, are available in the NHANES data release file. The selection of the correct sample weights for the analyses depends on the variables used. All the interview and MEC exam weights covered in this study are available in the demographic files. Since the MEC examined sample persons are a subset of those interviewed in the survey, we used the combined MEC exam weight for analysis. NHANES2015-2018 involved a combination of two survey cycles (four years), and the data were weighted according to the information that the NCHS provided analysts on how to combine multiple cycles and construct appropriate weights. MEC4YR = 1/2 × WTMEC2YR [28]. ... All data in the study were analysed using the statistical package R (http://www.r-project.org; the Version: 3.4.3, 2018–02-18) and Empower Stats (http://www.empowerstats.com; X&Y Solutions Inc).",
   "software": "R 3.4.3 and EmpowerStats (X&Y Solutions Inc)",
   "missing_data": "complete case for sleep duration, sleep timing and albumin; handling of missing covariates unstated (Table 1 keeps 'unknown' as a category for hypertension, high cholesterol and cancer)",
   "quote_missing": "[Study design] a total of 9,973 participants aged ≥ 20 years with a complete set of sleep-related data and albumin data were included in this study. [Covariates] including hypertension (yes, no, unknown), high cholesterol (yes, no, unknown), cancer or malignancy (yes, no, unknown)."
  },
  "model": {
   "family": "linear",
   "weighted": true,
   "quote": "[Abstract] Weighted data were calculated according to analytical guidelines. Linear regression models and smooth curve fitting were used to assess and describe the relationship between sleep duration and albumin. [Statistical analysis] Linear regression models were used to describe the relationship between sleep duration and albumin levels, and three models were constructed: Model 1, was not adjusted for any variables; Model 2, was adjusted for sex, age, race, marital status, and moderate work activity; Model 3, was adjusted for all the covariates (sex, age, race, marital status, moderate work activity, TP, ALT, AST, Cr, UACR, HS-CRP, GLU, BMI, hypertension, high cholesterol, cancer or malignancy)."
  },
  "unstated": [
   "Whether strata and PSUs (SDMVSTRA, SDMVPSU) were used for variance estimation (only weights are described)",
   "How missing covariate values were handled, and the number of participants in Model 3 (Fig. 1 excludes only on age, sleep and albumin data)",
   "Whether Cr and UACR entered the models log2-transformed (Table 1 shows log2 values)",
   "Which glucose 'GLU' is: refrigerated serum glucose from the biochemistry profile (LBDSGLSI) or fasting plasma glucose (LBDGLUSI, fasting subsample only)",
   "Coding of age, BMI, TP, ALT, AST, HS-CRP and GLU in the models (presumably continuous)",
   "How 'unknown' (refused or don't know) answers to hypertension, high cholesterol and cancer entered the model, and how refused/don't know answers to moderate work activity were handled",
   "Where Other Hispanic participants were placed among the four race groups",
   "Reference categories of the categorical covariates",
   "How the SLD012 top and bottom codes (2 = less than 3 hours, 14 = 14 hours or more in 2017-2018) were treated",
   "Whether sleep duration and sleep timing were modelled separately (the Table 2 footnote lists neither as a covariate of the other)",
   "Whether pregnant women were excluded (no such exclusion is described)"
  ],
  "notes": "Headline rule: the abstract's Results open with the inverted U and the 7.5 h peak, which carry no estimate; the first coding with an estimate is the categorical sleep duration with a middle reference (7-8 h), for which the clarified rule keeps the first contrast the abstract reports, <= 5 h vs 7-8 h (Model 3 of Table 2), significant, so it is the headline. The abstract's second estimate (> 9 h vs 7-8 h, -0.48 (-0.68, -0.27)) is in other_estimates. Cycles: 2015-2016 + 2017-2018 only; no combination with the overlapping 2017-March 2020 pre-pandemic files. Exposure comparability: SLD012 was a direct question in 2015-2016 but is derived from usual sleep and wake times (SLQ300/SLQ310) in 2017-2018 and 2021-2023, where the two cycles' variables are identical; only weekday/workday sleep is used (weekend hours SLD013 exist only from 2017-2018). Outcome comparability: albumin by bromcresol purple on Cobas 8000 in 2021-2023, no bridging adjustment recommended. Model 3 includes moderate work activity (PAQ620), which 2021-2023 does not have. Fitted in R 3.4.3 with EmpowerStats. The MDRD eGFR is used only for subgroup analyses (Table 4), not as a covariate. Subgroup text errors (not the headline): the Results text reuses the non-Hispanic white estimates (<= 5 h: -1.67 (-2.16, -1.17); > 9 h: -0.65 (-1.00, -0.30)) for the married subgroup, and the non-Hispanic white > 9 h estimate for the high-cholesterol subgroup, whereas Table 4 gives -1.38 (-1.75, -1.01) and -0.69 (-0.97, -0.40) for married and -0.83 (-1.15, -0.50) for high cholesterol at > 9 h. Supplementary Tables S1 (univariate) and S2 (subgroups) were not needed and not fetched. Fig. 1 was read from the publisher PDF (dl/supp/row284/).",
  "adjudication": null
 },
 {
  "id": "row194",
  "rank": 149,
  "row": 194,
  "doi": "10.1186/s12872-023-03333-5",
  "pmcid": "PMC10258976",
  "title": "The association between serum uric acid and creatine phosphokinase in the general population: NHANES 2015–2018",
  "authors": [
   "Chen, Xinxin",
   "You, Jiuhong",
   "Zhou, Mei",
   "Ma, Hui",
   "Huang, Cheng"
  ],
  "year": 2023,
  "journal": "BMC Cardiovascular Disorders",
  "table_a": {
   "predictor": "Serum uric acid levels",
   "condition": "Creatine phosphokinase",
   "population": "US population"
  },
  "headline": {
   "abstract_quote": null,
   "table_location": "Table 2 ('The association between serum uric acid (μmol/L) and creatine phosphokinase (IU/L)'), row 'Serum uric acid (μmol/L)', column 'Model 3β (95% CI), p value'. Chosen by step 2 of the headline rule, since the abstract reports no estimate.",
   "table_quote": "| Serum uric acid (μmol/L) | 0.30(0.27, 0.33), < 0.001 | 0.14 (0.11, 0.17), < 0.001 | 0.12 (0.09, 0.15), < 0.001 |",
   "measure": "beta",
   "estimate": 0.12,
   "ci_low": 0.09,
   "ci_high": 0.15,
   "p_value": "< 0.001",
   "exposure_contrast": "per 1 μmol/L higher serum uric acid (continuous); beta is the difference in serum CPK in IU/L (CPK untransformed)",
   "model_label": "Model 3",
   "covariates_in_this_model": [
    "Age",
    "sex",
    "race/ethnicity",
    "education level",
    "income poverty ratio",
    "BMI",
    "waist circumference",
    "physical activity",
    "smoking behavior",
    "alcohol behavior",
    "alanine aminotransferase",
    "aspartate aminotransferase",
    "high-density lipoprotein cholesterol",
    "total cholesterol",
    "triglycerides",
    "low-density lipoprotein cholesterol",
    "total protein",
    "hypertension",
    "diabetes",
    "weak/failing kidneys"
   ],
   "n_analytic": 8431,
   "n_quote": "[Abstract] Data from the National Health and Nutrition Examination Survey (NHANES) 2015–2018 were used, including a total of 8,431 subjects aged ≥ 30 years. [Study population] Ultimately, 8,431 subjects remained for the final analysis (Fig. 1). [Fig. 1 (image; transcribed in dl/supp/row194/transcribed_from_images.txt)] Participants included in the final analysis (n=8,431) [comment: Table 2 does not print the number of participants in each model; whether Model 3 lost participants with missing covariates is not stated.]",
   "events": null
  },
  "cycles": [
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=30",
   "defining": "US adults aged 30 years or older (the 'general population'); no condition or characteristic defines the population",
   "inclusion": "NHANES 2015-2018 participants aged >= 30 with serum uric acid and CPK measured and CPK <= 1000 U/L",
   "exclusions": [
    "aged < 30: 9,739 (19,225 to 9,486, Fig. 1)",
    "missing serum uric acid (n = 1,004) or CPK (n = 8): 1,012 (9,486 to 8,474, Fig. 1)",
    "CPK > 1000 U/L (rhabdomyolysis): 43 (8,474 to 8,431, Fig. 1)"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "[Study population] The study population was limited to participants aged ≥ 30 years with complete data on sUA and CPK. The exclusion criteria were those who have missing sUA data (n = 1,004) or CPK data (n = 8) and CPK > 1000 U/L (n = 43). Ultimately, 8,431 subjects remained for the final analysis (Fig. 1). [Discussion] We only excluded subjects with rhabdomyolysis (CPK > 1000 U/L).\n[Fig. 1 (image), transcribed box text, one box per line:]\nTotal participants from NHANES 2015-2018 (n=19,225)\nIndividuals aged ＜30 (n=9,739)\nIndividuals aged ≥30 (n=9,486)\nMissing data for serum uric acid or creatine phosphokinase (n=1,012)\nParticipants with complete data (n=8,474)\nIndividuals with creatine phosphokinase ＞1000 U/L (n=43)\nParticipants included in the final analysis (n=8,431)"
  },
  "exposure": {
   "definition": "Serum uric acid (sUA) in μmol/L from the standard biochemistry profile (Beckman DxC800 timed endpoint method in 2015-2016; Roche Cobas 6000 c501 in 2017-2018), continuous.",
   "nhanes_variables": [
    "LBDSUASI (μmol/L; LBXSUA is the same in mg/dL)"
   ],
   "nhanes_files": [
    "BIOPRO (BIOPRO_I, BIOPRO_J; BIOPRO_L in 2021-2023)"
   ],
   "transform": "none (per 1 μmol/L); quartiles only in Table 1 and in the Table 3 adjusted means",
   "categories": "[comment: quartiles Q1 to Q4 are used only for Table 1 and the Table 3 adjusted CPK means; cutpoints are not reported.] [Statistical analysis] sUA was shown as a continuous variable, and divided into quartiles.",
   "quote": "[Variables] The main variables of this study were sUA and CPK, which were the independent variable and dependent variable, respectively. sUA and CPK were both items in the standard biochemical examination and were measured at the same time. In the NHANES 2015–2016 cycle, sUA was measured by the DxC800 using a timed endpoint method, and CPK was measured by the DxC800 using an enzymatic rate method. In the NHANES 2016–2018 cycle, sUA and CPK were measured on a Roche Cobas 6000 (c501 module) analyzer, and the methods were based on the same principles used in the 2015–2016 cycle."
  },
  "outcome": {
   "definition": "Serum creatine phosphokinase (CPK) in IU/L from the standard biochemistry profile (DxC800 enzymatic rate method in 2015-2016; Roche Cobas 6000 c501 in 2017-2018), continuous and untransformed; participants with CPK > 1000 U/L excluded.",
   "nhanes_variables": [
    "LBXSCK"
   ],
   "nhanes_files": [
    "BIOPRO (BIOPRO_I, BIOPRO_J; BIOPRO_L in 2021-2023)"
   ],
   "quote": "[Variables] sUA and CPK were both items in the standard biochemical examination and were measured at the same time. [Table 2 title] The association between serum uric acid (μmol/L) and creatine phosphokinase (IU/L)"
  },
  "covariates": [
   {
    "name": "age",
    "coding": "unstated (Table 1 reports mean ± SD in years; presumably continuous)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Models 2 and 3."
   },
   {
    "name": "sex",
    "coding": "male, female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Models 2 and 3."
   },
   {
    "name": "race/ethnicity",
    "coding": "white, black, Mexican American, and other race/ethnicity",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Models 2 and 3. Variable not named; with four groups, where Other Hispanic went is unstated (presumably 'other')."
   },
   {
    "name": "education level",
    "coding": "less than high school, high school, and college graduate or above",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Model 3. Where 'some college or AA degree' went is unstated; Table 1 puts 64.1% in 'College graduate or above', which suggests it was merged into the top group."
   },
   {
    "name": "income-poverty ratio",
    "coding": "continuous (Table 1 mean ± SD)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "BMI",
    "coding": "under/normal weight < 25, overweight >= 25 and < 30, obesity >= 30 kg/m2 (Table 1); whether the model used the categories or continuous BMI is unstated",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "waist circumference",
    "coding": "continuous, cm",
    "nhanes_variables": [
     "BMXWAIST"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "physical activity",
    "coding": "yes or no to moderate-intensity sports, fitness, or recreational activities for at least 10 min continuously in a typical week",
    "nhanes_variables": [
     "PAQ665"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3. Approximate in 2021-2023: PAQ_L has no yes/no item but asks how often the person does moderate leisure-time activity (PAD790Q/PAD790U), so 'any' = frequency > 0; the 10-minute minimum and 'typical week' framing are gone."
   },
   {
    "name": "smoking behavior",
    "coding": "yes or no to 'Have you smoked at least 100 cigarettes in your life?'",
    "nhanes_variables": [
     "SMQ020"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "alcohol behavior",
    "coding": "none or rarely, sometimes (1 to 3 times a month), often (>= 1 time a week); Table 1 also has a 'Missing data' group (26.7%)",
    "nhanes_variables": [
     "ALQ120Q/ALQ120U (2015-2016)",
     "ALQ121 (2017-2018, 2021-2023)"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3. The mapping from NHANES frequencies to the three groups, and the place of never drinkers, are unstated."
   },
   {
    "name": "alanine aminotransferase (ALT)",
    "coding": "continuous, U/L",
    "nhanes_variables": [
     "LBXSATSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Model 3. The CDC recommends a Deming equation to bridge Cobas 6000 and Cobas 8000 ALT."
   },
   {
    "name": "aspartate aminotransferase (AST)",
    "coding": "continuous, U/L",
    "nhanes_variables": [
     "LBXSASSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "high-density lipoprotein cholesterol (HDL-C)",
    "coding": "continuous, mmol/L",
    "nhanes_variables": [
     "LBDHDDSI"
    ],
    "nhanes_files": [
     "HDL"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "total cholesterol (TC)",
    "coding": "continuous, mmol/L",
    "nhanes_variables": [
     "LBDTCSI (TCHOL) or LBDSCHSI (BIOPRO): unstated which"
    ],
    "nhanes_files": [
     "TCHOL or BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "triglycerides (TG)",
    "coding": "continuous, mmol/L",
    "nhanes_variables": [
     "LBDTRSI (fasting, TRIGLY) or LBDSTRSI (BIOPRO): unstated which"
    ],
    "nhanes_files": [
     "TRIGLY or BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "low-density lipoprotein cholesterol (LDL-C)",
    "coding": "continuous, mmol/L",
    "nhanes_variables": [
     "LBDLDLSI (Friedewald)"
    ],
    "nhanes_files": [
     "TRIGLY"
    ],
    "in_2021_2023": true,
    "note": "Model 3. Exists only for the morning fasting subsample in every cycle, yet the analytic N (8,431) is far larger than that subsample, so missing LDL-C must have been handled somehow; the paper does not say how."
   },
   {
    "name": "total protein",
    "coding": "continuous, g/L",
    "nhanes_variables": [
     "LBDSTPSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "hypertension",
    "coding": "mean systolic BP >= 130 mmHg, mean diastolic BP >= 80 mmHg, ongoing use of antihypertensive drugs, or self-reported doctor-diagnosed hypertension",
    "nhanes_variables": [
     "BPXSY1-BPXSY4, BPXDI1-BPXDI4 (auscultatory, 2015-2018)",
     "BPXOSY1-3, BPXODI1-3 (oscillometric, 2021-2023)",
     "BPQ020",
     "BPQ050A (2015-2018) / BPQ150 (2021-2023)"
    ],
    "nhanes_files": [
     "BPX (BPXO_L in 2021-2023)",
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3. Device differs (oscillometric in 2021-2023), same construct and units."
   },
   {
    "name": "diabetes",
    "coding": "self-reported doctor-diagnosed diabetes or current use of insulin or diabetes medications",
    "nhanes_variables": [
     "DIQ010",
     "DIQ050",
     "DIQ070"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3. Table 1 percentages (13.6% yes, 83.7% no) do not sum to 100, so some third group (borderline or missing) existed; its handling is unstated."
   },
   {
    "name": "weak/failing kidneys",
    "coding": "self-reported physician-diagnosed weak/failing kidneys",
    "nhanes_variables": [
     "KIQ022"
    ],
    "nhanes_files": [
     "KIQ_U"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   }
  ],
  "design": {
   "weights": "unstated which weight ('NHANES sample weights were taken into account'); for these MEC laboratory variables over the two cycles the CDC-recommended weight would be WTMEC2YR/2, but the paper does not name one",
   "strata_psu": "unstated",
   "quote": "[Statistical analysis] All statistical analyses were performed using EmpowerStats (X&Y Solutions, Boston, MA) and R (version 3.4.3). The NHANES sample weights were taken into account when calculating the estimates.",
   "software": "EmpowerStats (X&Y Solutions, Boston, MA) and R (version 3.4.3)",
   "missing_data": "complete case for uric acid and CPK; covariate handling unstated except alcohol, which keeps a 'Missing data' category in Table 1 (26.7%)",
   "quote_missing": "[Study population] The study population was limited to participants aged ≥ 30 years with complete data on sUA and CPK. [Table 1, row under 'Alcohol behavior (%)' (columns Total, Q1, Q2, Q3, Q4)] | Missing data | 26.7 | 29.7 | 29.0 | 24.8 | 23.2 |  |"
  },
  "model": {
   "family": "linear",
   "weighted": true,
   "quote": "[Abstract] Weighted multiple regression analysis was used to estimate the independent relationship between sUA and CPK. [Statistical analysis] After adjusting for potential confounders, weighted multiple regression analysis was used to estimate the independent relationship between sUA and CPK. [Association between sUA and CPK] Three models were established: model 1, unadjusted; model 2, sex, age, and race were adjusted; and model 3, adjusted for covariates, as shown in Table 1."
  },
  "unstated": [
   "Which sample weight was used, and whether strata and PSUs (SDMVSTRA, SDMVPSU) were used",
   "How covariates with missing values were handled (only alcohol shows a 'Missing data' category) and the number of participants in Model 3",
   "Whether BMI entered as the three Table 1 categories or as continuous BMI",
   "Coding of age, income-poverty ratio, waist circumference, lipids, ALT, AST and total protein (presumably continuous)",
   "Whether TC and TG came from the lipid files (TCHOL, fasting TRIGLY) or the biochemistry profile (LBXSCH, LBXSTR), and how LDL-C, measured only in the fasting subsample, was handled for everyone else",
   "Where Other Hispanic participants were placed among the four race groups",
   "Where 'some college or AA degree' was placed among the three education groups",
   "How the alcohol groups were built from ALQ120Q/ALQ120U (2015-2016) and ALQ121 (2017-2018), and where never drinkers went",
   "Which blood pressure readings were averaged for the hypertension definition, and how antihypertensive drug use was identified",
   "How borderline diabetes and refused/don't know answers were classified",
   "Reference categories of the categorical covariates",
   "Whether pregnant women were excluded (no such exclusion is described)"
  ],
  "notes": "Headline from step 2: the abstract states the positive association without any estimate, so the Table 2 Model 3 estimate for continuous sUA is the headline. Cycles: 2015-2016 + 2017-2018 only; no combination with the overlapping 2017-March 2020 pre-pandemic files. Instruments: 2015-2016 Beckman DxC800, 2017-2018 Roche Cobas 6000, 2021-2023 Roche Cobas 8000 (CDC bridging: uric acid needs no adjustment; CPK has a recommended Deming equation, see E5). The outcome CPK is analysed untransformed in IU/L despite its skew (SD 120.7 around a mean of 139.0), and beta is per 1 μmol/L of uric acid, so beta per mg/dL would be about 59.48 times larger (1 mg/dL = 59.48 μmol/L). Fitted in EmpowerStats and R 3.4.3. Table 3 gives adjusted CPK means by sUA quartile within sex-by-race strata (quartile labels '3nd', '4nd' as printed); Table 4's two-piecewise model (turning point 428.3 μmol/L) is for females only and prints '0.10 (0.00, 0.10), 0.001' (a CI bound equal to the estimate, apparently from rounding). Fig. 1 was read from the publisher PDF (dl/supp/row194/).",
  "adjudication": null
 },
 {
  "id": "row303",
  "rank": 152,
  "row": 303,
  "doi": "10.3389/fendo.2023.1245199",
  "pmcid": "PMC10644783",
  "title": "Association between systemic immune-inflammation index and diabetes: a population-based study from the NHANES",
  "authors": [
   "Nie, Yiqi",
   "Zhou, Haiting",
   "Wang, Jing",
   "Kan, Hongxing"
  ],
  "year": 2023,
  "journal": "Frontiers in Endocrinology",
  "table_a": {
   "predictor": "Systemic immune-inflammation index",
   "condition": "Diabetes",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "Weighted multivariate regression analysis found that SII was positively associated with diabetes, and in model 3, this positive association remained stable (OR = 1.04; 95% CI: 1.02–1.06; p = 0.0006), indicating that each additional unit of SII, the possibility of having diabetes increased by 4%.",
   "table_location": "Table 3 ('Association between systemic immune-Inflammation index and diabetes.'), row 'SII/100', column 'Fully Adjusted Model (Model 3)', OR (95% CI) p-Value. The abstract's estimate matches Model 3 and the Results text.",
   "table_quote": "| SII/100 | 1.04 (1.02, 1.05) *** | 1.04 (1.02, 1.05) *** | 1.04 (1.02, 1.06) *** |",
   "measure": "OR",
   "estimate": 1.04,
   "ci_low": 1.02,
   "ci_high": 1.06,
   "p_value": "p = 0.0006 (abstract and Results text); Table 3 prints *** (footnote: *** p < 0.001)",
   "exposure_contrast": "per 100-unit increase in SII (SII/100), where SII = platelet count x neutrophil count / lymphocyte count with counts in 1000 cells/uL; the abstract's 'each additional unit of SII' is per unit of SII/100 according to Table 3 and the Results",
   "model_label": "Model 3 (Fully Adjusted Model)",
   "covariates_in_this_model": [
    "age",
    "sex",
    "race",
    "marital status",
    "income to poverty ratio",
    "education level",
    "drink",
    "smoke",
    "BMI",
    "high blood pressure",
    "blood urea nitrogen",
    "White blood cell count",
    "Platelet count",
    "chloride",
    "protein",
    "carbohydrate",
    "total sugars",
    "total fat",
    "cholesterol",
    "vigorous work activity"
   ],
   "n_analytic": 7877,
   "n_quote": "[Abstract] There were 7877 subjects in this study [Population research] 7877 participants were ultimately included in the study. [Figure 1 (image; transcribed in dl/supp/row303/transcribed_from_images.txt)] Final participants ( N= 7,877 ) [comment: Table 3 does not print the number of participants in each model; whether Model 3 lost participants with missing covariates (including dietary recalls) is not stated.]",
   "events": 1266
  },
  "cycles": [
   "2017-March 2020 (pre-pandemic)"
  ],
  "population": {
   "age": ">=20",
   "defining": "US adults aged 20 years or older; no condition or characteristic defines the population",
   "inclusion": "NHANES 2017-March 2020 participants aged 20 or older with SII and diabetes data",
   "exclusions": [
    "age < 20: 6,328 (15,560 to 9,232, Figure 1)",
    "missing SII data: 1,111 (9,232 to 8,121, Figure 1)",
    "missing diabetes data: 244 (8,121 to 7,877, Figure 1)"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "[Population research] In total, 15,560 participants from the 2017–2018 and 2019–2020 research years were included in this study, whereas 6,328 participants under the age of 20 were omitted. Following the exclusion of 1111 people with missing SII data and 244 participants with missing diabetes data, 7877 participants were ultimately included in the study.\n[Figure 1 (image), transcribed box text, one box per line:]\nParticipants from NHANES 2017 - 2020 ( N= 15,560 )\nAge < 20 ( N= 6,328 )\nParticipants with age>19 ( N= 9,232 )\nSII data missing ( N= 1,111 )\nParticipants with SII data ( N= 8,121 )\nDiabetes data missing ( N= 244 )\nFinal participants ( N= 7,877 )"
  },
  "exposure": {
   "definition": "Systemic immune-inflammation index SII = platelet count x neutrophil count / lymphocyte count, from the CBC (Beckman Coulter DxH 800), counts in 1000 cells/uL; entered as SII/100. The Methods sentence garbles the formula ('dividing the lymphocyte count by the platelet count, and then multiplying by the neutrophil count'); the Introduction gives the standard formula, and the mean SII of 524.91 fits it.",
   "nhanes_variables": [
    "LBXPLTSI",
    "LBDNENO",
    "LBDLYMNO"
   ],
   "nhanes_files": [
    "CBC (P_CBC; CBC_L in 2021-2023)"
   ],
   "transform": "per 100 units (SII/100); quartiles of SII/100 (Quartiles 1 reference) in Table 3",
   "categories": "[comment: quartiles of SII/100 in Table 3, Quartiles 1 = Reference; cutpoints not reported. Quartile sizes, Table 2 header row:] | N=1969 | N=1969 | N=1969 | N=1970 |",
   "quote": "[Introduction] Systemic Immunity-Inflammation Index (SII) is the platelet count multiplied by the neutrophil count divided by the lymphocyte count [Exposure variables and outcome variables] SII is a composite index made up of platelet count, neutrophil count, and lymphocyte count that is used as an exposure variable and is assessed using an automated hematology analysis instrument (CoulterDxH 800 analyzer). calculated by dividing the lymphocyte count by the platelet count, and then multiplying by the neutrophil count (19). [Relationship between SII and diabetes] Since the effect size was not obvious, we magnified the value of SII by 100 times to compare the relationship between SII/100 and diabetes."
  },
  "outcome": {
   "definition": "Self-reported doctor-diagnosed diabetes: a 'yes' answer to whether a doctor ever said the participant had diabetes (DIQ010 = 1). How 'borderline' (DIQ010 = 3) and refused/don't know answers were treated is unstated; 244 participants with missing diabetes data were excluded.",
   "nhanes_variables": [
    "DIQ010"
   ],
   "nhanes_files": [
    "DIQ (P_DIQ; DIQ_L in 2021-2023)"
   ],
   "quote": "[Exposure variables and outcome variables] Did your doctor inform you that you had diabetes? Are these other questions asked by a professional interviewer utilizing a computer-assisted personal interview (CAPI) system at home? The subject was deemed to have diabetes if his response was yes. Having diabetes diagnosed by a doctor was intended to be an outcome variable (20)."
  },
  "covariates": [
   {
    "name": "age",
    "coding": "continuous, years (Table 1 mean ± SD)",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Models 2 and 3."
   },
   {
    "name": "sex",
    "coding": "male, female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Models 2 and 3."
   },
   {
    "name": "race",
    "coding": "Mexican American, Non-Hispanic White, Non-Hispanic Black, Other Hispanic, Other Race (Table 1)",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Models 2 and 3."
   },
   {
    "name": "marital status",
    "coding": "Methods: married with a partner, divorced and separated, widowed, never married; Table 1: Married/Living with partner, Widowed/Divorced/Separated, Never married",
    "nhanes_variables": [
     "DMDMARTZ"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Model 3. P_DEMO and DEMO_L both carry the three-group DMDMARTZ, matching Table 1."
   },
   {
    "name": "income to poverty ratio",
    "coding": "0-1.5, 1.5-3.5, >3.5",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Model 3. Table 1's PIR percentages for the two upper groups repeat the BMI rows (typesetting error)."
   },
   {
    "name": "education level",
    "coding": "Methods: no high school education, high school education, high school education or above; Table 1: Less than high school, High school, More than high school",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "drink",
    "coding": "unstated (appears only in the Table 3 footnote)",
    "nhanes_variables": [
     "ALQ (item unstated)"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3 per the Table 3 footnote; not in the Methods' Model 3 list. ALQ_L asks the same alcohol items as P_ALQ, so any definition built from them is constructible."
   },
   {
    "name": "smoke",
    "coding": "smoked 100 cigarettes in life: yes, no",
    "nhanes_variables": [
     "SMQ020"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "BMI",
    "coding": "0-25, 25-30, >30 kg/m2 (normal weight, overweight, obese)",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "high blood pressure",
    "coding": "self-reported: whether a doctor said the participant had high blood pressure (yes, no)",
    "nhanes_variables": [
     "BPQ020"
    ],
    "nhanes_files": [
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "blood urea nitrogen",
    "coding": "continuous, mmol/L",
    "nhanes_variables": [
     "LBDSBUSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Model 3. The CDC recommends a Deming equation to bridge Cobas 6000 and Cobas 8000 BUN."
   },
   {
    "name": "White blood cell count",
    "coding": "continuous, 1000 cells/uL",
    "nhanes_variables": [
     "LBXWBCSI"
    ],
    "nhanes_files": [
     "CBC"
    ],
    "in_2021_2023": true,
    "note": "Model 3 per the Table 3 footnote; listed among covariates in the Methods but not in its Model 3 sentence."
   },
   {
    "name": "Platelet count",
    "coding": "continuous, 1000 cells/uL",
    "nhanes_variables": [
     "LBXPLTSI"
    ],
    "nhanes_files": [
     "CBC"
    ],
    "in_2021_2023": true,
    "note": "Model 3 per the Table 3 footnote; platelets are also a component of SII."
   },
   {
    "name": "chloride",
    "coding": "continuous, mmol/L",
    "nhanes_variables": [
     "LBXSCLSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "Model 3. The CDC recommends a Deming equation to bridge Cobas 6000 and Cobas 8000 chloride."
   },
   {
    "name": "protein (dietary)",
    "coding": "day-1 total intake, gm",
    "nhanes_variables": [
     "DR1TPROT"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "carbohydrate (dietary)",
    "coding": "day-1 total intake, gm",
    "nhanes_variables": [
     "DR1TCARB"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "total sugars (dietary)",
    "coding": "day-1 total intake, gm",
    "nhanes_variables": [
     "DR1TSUGR"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "total fat (dietary)",
    "coding": "day-1 total intake, gm",
    "nhanes_variables": [
     "DR1TTFAT"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "cholesterol (dietary)",
    "coding": "day-1 total intake, mg",
    "nhanes_variables": [
     "DR1TCHOL"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "Model 3."
   },
   {
    "name": "vigorous work activity (the Methods' 'whether you exercise regularly')",
    "coding": "yes, no (Table 1)",
    "nhanes_variables": [
     "PAQ605"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": false,
    "note": "Model 3 per the Table 3 footnote. PAQ_L asks only about leisure-time activity; there are no work-activity questions in 2021-2023."
   },
   {
    "name": "Albumin, refrigerated serum",
    "coding": "continuous, g/L",
    "nhanes_variables": [
     "LBDSALSI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "In the Methods' Model 3 list ('refrigerated serum content') and Covariates, but not in the Table 3 footnote."
   },
   {
    "name": "urine albumin content",
    "coding": "unstated",
    "nhanes_variables": [
     "URXUMA"
    ],
    "nhanes_files": [
     "ALB_CR"
    ],
    "in_2021_2023": true,
    "note": "Only in the Methods' Model 3 sentence; not in the Covariates paragraph or the Table 3 footnote."
   }
  ],
  "design": {
   "weights": "unstated which weight ('We employ a weighting approach'); Tables 1 and 2 are labelled 'weighted'. The pre-pandemic MEC weight is WTMECPRP and the day-1 dietary weight WTDRD1PP; the paper names neither.",
   "strata_psu": "unstated",
   "quote": "[Statistical methods] R Studio (version 4.2.2) and empowered stats (version 2.0) were used to do the statistical study. 0.05 was the threshold for significance. We employ a weighting approach to lower the dataset’s volatility. [Table 1 title] Baseline characteristics of study population according to Diabetes, weighted.",
   "software": "R Studio (version 4.2.2) and EmpowerStats (version 2.0)",
   "missing_data": "complete case for SII and diabetes; handling of missing covariates unstated (Figure 1 has no covariate exclusions)",
   "quote_missing": "[Abstract] After removing missing data for SII and diabetes, we examined patients older than 20 years."
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "[Abstract] Simultaneously, the relationship between SII and diabetes was examined using weighted multivariate regression analysis, subgroup analysis, and smooth curve fitting. [Statistical methods] To investigate the relationship between SII and the prevalence of diabetes, multivariate logistic regression was performed. Model 1 had no covariates, model 2 added covariates age, sex, and race, ... SII and diabetes were assessed in the model using odds ratios (ORs) and 95% confidence intervals (CIs)."
  },
  "unstated": [
   "Which weight (WTMECPRP, or the day-1 dietary weight WTDRD1PP since Model 3 includes day-1 intakes) and whether strata and PSUs were used",
   "The exact Model 3 covariate set: the Methods' list (urine albumin, serum albumin, 'whether you exercise regularly', no drink, no WBC, no platelets) and the Table 3 footnote (drink, WBC, platelet count, vigorous work activity, no albumin) disagree; the smooth-curve adjustment list in the Results is a third version",
   "How 'drink' was coded",
   "How DIQ010 'borderline' (3), refused (7) and don't know (9) answers were classified (the 244 excluded for missing diabetes data may include some of them)",
   "How missing covariates, including missing day-1 dietary recalls, were handled, and the number of participants in Model 3",
   "Coding of age, BUN, chloride, WBC, platelets and dietary intakes (presumably continuous)",
   "Reference categories of the categorical covariates",
   "SII/100 quartile cutpoints",
   "Whether pregnant women were excluded (no such exclusion is described)"
  ],
  "notes": "Headline: abstract, Results text and Table 3 Model 3 agree on OR 1.04 (1.02, 1.06); the exposure unit is SII/100 even though the abstract says 'each additional unit of SII'. Cycles: 2017-March 2020 pre-pandemic files only; not combined with the overlapping 2017-2018 files. Model 3 (Table 3 footnote) adjusts for platelet count, which is part of SII, and for white blood cell count. Vigorous work activity (PAQ605) is not in 2021-2023; everything else is. The outcome is self-reported diagnosis only (no HbA1c, glucose or medication criteria). Fitted with R Studio 4.2.2 and EmpowerStats 2.0. Figure 1 was read from the publisher PDF (dl/supp/row303/). The smooth-curve adjustment in the Results ('age, sex, education, marital status, poverty-income ratio, BMI, hypertension, smoking, regular exercise, albumin level, blood urea nitrogen level, chloride level, dietary carbohydrate intake, dietary total sugar intake, and dietary cholesterol intake, and high blood pressure') omits race, unlike Model 3.",
  "adjudication": null
 },
 {
  "id": "row236",
  "rank": 164,
  "row": 236,
  "doi": "10.1186/s12944-024-02306-7",
  "pmcid": "PMC11415990",
  "title": "Association between triglyceride glucose body mass index and urinary incontinence: a cross-sectional study from the National Health and Nutrition Examination Survey (NHANES) 2001 to 2018",
  "authors": [
   "Li, JiHang",
   "Xie, Ruijie",
   "Tian, Hu",
   "Wang, Dong",
   "Mo, MingShen",
   "Yang, JianKun",
   "Guo, WenBin"
  ],
  "year": 2024,
  "journal": "Lipids in Health and Disease",
  "table_a": {
   "predictor": "Triglyceride glucose body mass index",
   "condition": "Urinary incontinence",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "Considering all possible confounding variables, Multivariate logistic regression analysis showed a substantial relationship between elevated TyG-BMI values and a greater likelihood across all UI categories. Specifically, stratifying the TyG-BMI into quartiles revealed a pronounced positive correlation in the top quartile relative to the bottom, reflected in increased odds ratios for SUI, UUI, and MUI (SUI: OR = 2.36, 95% CI 2.03–2.78, P < 0.0001; UUI: OR = 1.86, 95% CI 1.65–2.09, P < 0.0001; MUI: OR = 2.07, 95% CI 1.71–2.51, P < 0.0001).",
   "table_location": "Table 2 (Associations between TyG-BMI and urinary incontinence), SUI block, row 'Q4' (reference row 'Q1'), column 'Model 3' under 'OR (95% CI), P-value'",
   "table_quote": "| Q4 | 1.86 (1.63,2.11), < 0.0001 | 2.46 (2.14,2.84), < 0.0001 | 2.36 (2.03,2.78), < 0.0001 |",
   "measure": "OR",
   "estimate": 2.36,
   "ci_low": 2.03,
   "ci_high": 2.78,
   "p_value": "< 0.0001",
   "exposure_contrast": "TyG-BMI quartile 4 (287.23-679.46) vs quartile 1 (112.58-204.87); outcome SUI (stress urinary incontinence) yes vs no",
   "model_label": "Model 3 (fully adjusted)",
   "covariates_in_this_model": [
    "age",
    "gender",
    "race/ethnicity",
    "education level",
    "marital status",
    "the family poverty ratio",
    "smoking status",
    "alcohol use",
    "vigorous activity",
    "moderate activity",
    "diabetes",
    "hypertension",
    "high cholesterol",
    "health insurance coverage"
   ],
   "n_analytic": 18751,
   "n_quote": "Methods, Data source and participants: \"Ultimately, the study comprised 18,751 participants (Fig. 1).\" Abstract, Results: \"A thorough investigation was conducted with 18,751 subjects\" Table 1 header: \"N = 18,751\"",
   "events": 4352
  },
  "cycles": [
   "2001-2002",
   "2003-2004",
   "2005-2006",
   "2007-2008",
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=20",
   "defining": "US adults aged 20 and over; no sex or condition restriction",
   "inclusion": "NHANES 2001-2018 participants aged 20+ with complete UI questionnaire data and BMI, triglyceride and fasting glucose data, not pregnant (n = 18,751)",
   "exclusions": [
    "Under 20 years (n = 41,150)",
    "Incomplete UI data (n = 6,365)",
    "Lacking BMI, triglyceride, and fasting glucose data (n = 24,593)",
    "Pregnant (n = 492)"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "Methods, Data source and participants: \"Initially, 91,351 individuals were recruited. Exclusions included participants under 20 years (n = 41,150), those with incomplete UI data (n = 6,365), those lacking BMI, triglyceride, and fasting glucose data (n = 24,593), and pregnant individuals (n = 492). Ultimately, the study comprised 18,751 participants (Fig. 1).\" Abstract, Methods: \"adults 20 years and older with UI were included in cross-sectional research using the data obtained from the National Health and Nutrition Examination Survey (NHANES) from 2001 to 2018\""
  },
  "exposure": {
   "definition": "TyG-BMI = ln[triglycerides (mg/dL) x fasting blood glucose (mg/dL) / 2] x BMI (kg/m2), natural log; categorized into quartiles of the analytic sample (cutpoints below; Table 1's near-equal quartile counts suggest unweighted quartiles). Headline: Q4 vs Q1.",
   "nhanes_variables": [
    "LBXTR",
    "LBXTLG",
    "LBXGLU",
    "BMXBMI"
   ],
   "nhanes_files": [
    "TRIGLY",
    "GLU",
    "BMX"
   ],
   "transform": "quartiles (Q4 vs Q1 for the headline); the per-unit continuous estimate is in Table 2 but not the headline",
   "categories": "Results, Characteristics of the participants: \"The designated quartiles for triglyceride-glucose body mass index were Q1 (112.58–204.87), Q2 (204.87–242.54), Q3 (242.54–287.23), and Q4 (287.23–679.46).\" Table 1 header (quartile sizes): \"| N = 4688 | N = 4687 | N = 4688 | N = 4688 |\"",
   "quote": "Methods, Exposure and outcome definitions: \"The exposure variable was defined as TyG-BMI, which was computed by the following formula: Ln [triglycerides (mg/dl) * fasting blood glucose (mg/dl)/2] * BMI (kg/m2) [17]. Triglycerides were enzymatically measured using the spectrophotometric method (Roche Cobas 6000 chemical analyzer), while plasma glucose concentrations were determined using the glucose-oxidase method (Roche/Hitachi Cobas C 501 chemical analyzer).\" Methods, Statistical analysis: \"Additionally, TyG-BMI was categorized into quartiles from a continuous measure, enabling a trend analysis to identify potential correlations with UI.\""
  },
  "outcome": {
   "definition": "Stress urinary incontinence (SUI): self-reported involuntary leakage of a small amount of urine with physical activity such as coughing or exercise in the past 12 months, yes vs no. The comparison group is everyone without SUI (Table 1: SUI No 14,399 + Yes 4,352 = 18,751), which includes people with urge leakage only. Whether people with both stress and urge leakage (MUI) also count as SUI is not stated (see notes: the counts imply they do).",
   "nhanes_variables": [
    "KIQ042"
   ],
   "nhanes_files": [
    "KIQ_U"
   ],
   "quote": "Methods, Exposure and outcome definitions: \"By assessing the NHANES data, the incidence of UI was determined in the current research, focusing on two questionnaire items related to bladder function. SUI was identified in participants reporting involuntary urine leakage in small amounts due to physical activities, including coughing or exercise, within the past year. UUI was distinguished by the concurrent presence of involuntary urine discharge and an urgent, compelling urge to urinate during the same period. Individuals reporting symptoms of both SUI and UUI were categorized as having MUI, offering a comprehensive classification of UI subtypes in the cohort.\""
  },
  "covariates": [
   {
    "name": "age",
    "coding": "years; Table 1 reports mean ± SD; form in the model (continuous or grouped) not stated",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "gender",
    "coding": "Male, Female (Table 1)",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "race/ethnicity",
    "coding": "Mexican American, Other Hispanic, Non-Hispanic White, Non-Hispanic Black, Other Races (Table 1)",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "education level",
    "coding": "under high school, high school, above high school (Methods); Table 1: Less than high school, High school or GED, Above high school",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DMDEDUC2 levels 1-2 (less than 9th grade; 9-11th grade) map to under high school, 3 to high school or GED, 4-5 to above high school; the mapping is not stated."
   },
   {
    "name": "marital status",
    "coding": "Methods: single, married, living together, widowed, divorced, or separated; Table 1: Married or living with partners; Widowed, divorced, or separated; Never married",
    "nhanes_variables": [
     "DMDMARTL",
     "DMDMARTZ"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "2021-2023 has DMDMARTZ with exactly Table 1's three groups; earlier cycles used DMDMARTL (six levels)."
   },
   {
    "name": "the family poverty ratio",
    "coding": "income-to-poverty ratio grouped < 1.3, 1.3-3.5, > 3.5 (Methods; Table 1 prints the top group as >= 3.5)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Whether entered grouped or continuous is not stated; Table 1 shows it grouped; its three groups sum to 17,278 of 18,751 (5078 + 6717 + 5483), so 1,473 lacked PIR before imputation."
   },
   {
    "name": "smoking status",
    "coding": "never; current; former (at least 100 cigarettes in lifetime, not smoking now). Table 1: Never, Past, Current",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "alcohol use",
    "coding": "at least 12 alcoholic drinks per year vs non-consumers (Table 1: Yes, No)",
    "nhanes_variables": [
     "ALQ101"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": false,
    "note": "The paper's wording matches ALQ101 (at least 12 drinks of any alcoholic beverage in any one year), which 2021-2023 does not ask (nor did 2017-2018, so how 2017-2018 was coded is not stated). 2021-2023 can approximate drinks in the past 12 months from ALQ121 (frequency) and ALQ130 (drinks per drinking day), or use ALQ111 (ever drank)."
   },
   {
    "name": "vigorous activity",
    "coding": "Yes/No (Table 1 'Vigorous physical activity'); questions and domain (work, transport, leisure) not stated",
    "nhanes_variables": [
     "PAQ650",
     "PAD200"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": true,
    "note": "2021-2023 asks only leisure-time activity: PAD810Q/PAD810U (frequency of vigorous leisure activity). Constructible as leisure-time vigorous activity yes/no; not if the paper also counted work activity (e.g. PAQ605)."
   },
   {
    "name": "moderate activity",
    "coding": "Yes/No (Table 1 'Moderate physical activity'); questions and domain not stated",
    "nhanes_variables": [
     "PAQ665",
     "PAD320"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": true,
    "note": "2021-2023: PAD790Q/PAD790U (frequency of moderate leisure activity). Same caveat as vigorous activity."
   },
   {
    "name": "diabetes",
    "coding": "physician diagnosis, anti-diabetic medication, or HbA1c > 6.5% (Yes/No)",
    "nhanes_variables": [
     "DIQ010",
     "DIQ050",
     "DIQ070",
     "LBXGH"
    ],
    "nhanes_files": [
     "DIQ",
     "GHB"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "hypertension",
    "coding": "physician diagnosis, antihypertensive medication, or blood pressure > 140/90 mmHg (Yes/No)",
    "nhanes_variables": [
     "BPQ020",
     "BPQ050A",
     "BPQ150",
     "BPXSY1",
     "BPXSY2",
     "BPXSY3",
     "BPXSY4",
     "BPXDI1",
     "BPXDI2",
     "BPXDI3",
     "BPXDI4",
     "BPXOSY1",
     "BPXOSY2",
     "BPXOSY3",
     "BPXODI1",
     "BPXODI2",
     "BPXODI3"
    ],
    "nhanes_files": [
     "BPQ",
     "BPX",
     "BPXO"
    ],
    "in_2021_2023": true,
    "note": "2021-2023: BPQ150 (taking high blood pressure medication) replaces BPQ050A, and oscillometric BPXO readings replace auscultatory BPX (same construct and units). Which readings were averaged is not stated."
   },
   {
    "name": "high cholesterol",
    "coding": "physician diagnosis, cholesterol-lowering medication, or total cholesterol >= 240 mg/dL (Yes/No)",
    "nhanes_variables": [
     "BPQ080",
     "BPQ100D",
     "BPQ101D",
     "LBXTC"
    ],
    "nhanes_files": [
     "BPQ",
     "TCHOL"
    ],
    "in_2021_2023": true,
    "note": "2021-2023: BPQ101D (taking meds to lower blood cholesterol) replaces BPQ100D. The paper's sentence is garbled ('a physician diagnosed participants used hypercholesterolemia medication')."
   },
   {
    "name": "health insurance coverage",
    "coding": "Yes/No (Table 1)",
    "nhanes_variables": [
     "HIQ011"
    ],
    "nhanes_files": [
     "HIQ"
    ],
    "in_2021_2023": true,
    "note": ""
   }
  ],
  "design": {
   "weights": "NHANES sample weights; which weight (fasting subsample WTSAF2YR, MEC WTMEC2YR, or interview) and how it was rescaled for nine combined cycles are not stated. The exposure needs fasting labs, so the fasting subsample weight would be the standard choice.",
   "strata_psu": "not stated",
   "quote": "Methods, Statistical analysis: \"Statistical analyses followed CDC guidelines and included NHANES sample weights.\" Methods, Data source and participants: \"We also applied sample weighting and multiple imputation methods in our analysis to further enhance the reliability of the results.\"",
   "software": "R 3.4.3 and EmpowerStats (Empower software, X&Y Solutions)",
   "missing_data": "imputation (multiple imputation; variables imputed, method, number of imputations and pooling not stated). Participants missing UI, BMI, triglyceride or fasting glucose were excluded first, so the imputation concerns covariates.",
   "quote_missing": "Methods, Data source and participants: \"To deal with missing values in the data, we used multiple imputation method.\""
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "Results, Correlation of TyG-BMI with the likelihood of urinary incontinence: \"Weighted multivariate logistic regression was used, adjusting models in a tiered approach: unadjusted (Model 1), moderately adjusted (Model 2), and fully adjusted (Model 3) for a precise assessment of the relationship.\" Methods, Statistical analysis: \"Statistical analyses were performed using R software version 3.4.3 from The R Foundation and Empower software (X&Y Solutions, Inc., Boston, MA, USA).\""
  },
  "unstated": [
   "Which sample weight (fasting subsample, MEC or interview) and how weights were combined over nine cycles (including 2001-2002's 4-year weights)",
   "Whether strata and PSUs were used for variance estimation",
   "Multiple imputation details: variables imputed, method, number of imputations, pooling",
   "Whether participants with both stress and urge leakage (MUI) are counted in SUI; the reference group for SUI (Table 1 implies everyone without SUI)",
   "How Model 3 enters age, PIR and the other covariates (continuous or grouped), and the reference categories",
   "Which physical activity questions and domains define vigorous and moderate activity",
   "How alcohol use was coded in 2017-2018, when the 12-drinks-per-year question (ALQ101) was not asked",
   "Which blood pressure readings define hypertension (> 140/90 as printed, average of which readings)",
   "Whether quartile cutpoints were weighted (Table 1's near-equal counts suggest unweighted sample quartiles)",
   "Fasting requirements or exclusions applied to the triglyceride and glucose values"
  ],
  "notes": "Headline choice: Suchak et al. label the condition 'Urinary incontinence', but the paper models three UI types and never any-UI; the abstract lists SUI first, so the headline is SUI, Q4 vs Q1, Model 3 (OR 2.36, 2.03-2.78). The next abstract estimates are UUI (OR 1.86, 1.65-2.09, KIQ044) and MUI (OR 2.07, 1.71-2.51, KIQ042 and KIQ044). Outcome items: SUI uses KIQ042 only; UUI uses KIQ044; MUI both; KIQ046 is never used, so the 2021-2023 gap in KIQ046 does not affect any of the three. Overlap inference: Table 1 counts (SUI 4352, UUI 4264, MUI 1970) fit overlapping definitions (any leakage = 4352 + 4264 - 1970 = 6646, 35% of 18,751); if SUI and UUI excluded MUI, the three exclusive types would cover 10,586 (56%), which is implausible. So SUI is most likely KIQ042 = yes regardless of urge leakage. Table 1 percentages are unweighted proportions despite the text saying weighted; the abstract's prevalences (23.59/19.42/9.32%) differ from Table 1's (23.21/22.74/10.51%). Table 1's SUI and UUI rows have identical Q2 counts (970 yes, 3717 no), a possible copy error. The ROC 'cut-off values' (1.164, 1.262, 2.193) are not on the TyG-BMI scale (112.58-679.46), so Table 3's thresholds cannot be applied as printed. Analyses used EmpowerStats with R 3.4.3. No correction or erratum was found. Cycles: 2001-2018 only (91,351 = sum of the nine 2001-2002 to 2017-2018 DEMO files); the 2017-March 2020 pre-pandemic files were not used, so there is no 2017-2018 overlap.",
  "adjudication": null
 },
 {
  "id": "row062",
  "rank": 184,
  "row": 62,
  "doi": "10.3389/fnut.2023.1218166",
  "pmcid": "PMC10552180",
  "title": "A cross-sectional study on the association between dietary inflammatory index and hyperuricemia based on NHANES 2005–2018",
  "authors": [
   "Wang, Hao",
   "Qin, Shengmei",
   "Li, Feng",
   "Zhang, Huanhuan",
   "Zeng, Ling"
  ],
  "year": 2023,
  "journal": "Frontiers in Nutrition",
  "table_a": {
   "predictor": "Dietary inflammatory index",
   "condition": "Hyperuricemia",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "After adjusting confounding factors, the odds of hyperuricemia are significantly higher in the second (OR 1.17, 95% CI 1.07–1.29) and third tertiles (OR 1.31, 95% CI 1.19–1.44) relative to the first one.",
   "table_location": "main regression table, fully adjusted model, row tertile 3",
   "table_quote": "| T2 (0.79, 2.60) | 1.15 (1.07, 1.23) | 1.14 (1.06, 1.23) | 1.17 (1.07, 1.29) |",
   "measure": "OR",
   "estimate": 1.31,
   "ci_low": 1.19,
   "ci_high": 1.44,
   "p_value": "not printed",
   "exposure_contrast": "DII tertile 3 vs tertile 1 (reference)",
   "model_label": "fully adjusted model",
   "covariates_in_this_model": [
    "age",
    "sex",
    "BMI",
    "race/ethnicity",
    "educational level",
    "smoking",
    "drinking",
    "MET",
    "eGFR",
    "diabetes",
    "hypertension",
    "hyperlipidemia"
   ],
   "n_analytic": 31781,
   "n_quote": "Abstract, Results: \"Among a total of 31,781 participants in the analysis, 5,491 had hyperuricemia.\" Table 3 block header: \"Total (n = 31,781)\" Methods, Statistical analysis (tertile sizes): \"T1 (−5.28 to 0.79, n = 10,594), T2 (0.79 to 2.60, n = 10,593), T3 (2.60 to 5.79, n = 10,594)\"",
   "events": 5491
  },
  "cycles": [
   "2005-2006",
   "2007-2008",
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": ">=18",
   "defining": "US adults aged 18 and over",
   "inclusion": "NHANES 2005-2018 participants aged 18+ with dietary data for DII and serum uric acid, not pregnant, with eGFR >= 60 mL/min/1.73 m2 (n = 31,781)",
   "exclusions": [
    "Without dietary data for DII calculation (n = 9,549)",
    "Missing uric acid data (n = 18,995)",
    "Under 18 years (n = 6,267)",
    "Pregnant (n = 642)",
    "Missing eGFR or eGFR < 60 mL/min/1.73 m2 (n = 2,956)"
   ],
   "exclusions_not_in_2021_2023": [
    "Pregnancy: 2021-2023 records pregnancy (RIDEXPRG, URXPREG) only for females 20-44, so pregnant 18-19-year-olds cannot be identified"
   ],
   "quote": "Methods, Study population: \"Exclusion criteria were as follows: (a) participants without dietary data for DII calculation (n = 9,549), (b) participants with missing uric acid data (n = 18,995), (c) individuals under the age of 18 years old (n = 6,267), (d) pregnant individuals (n = 642), and (e) individuals with missing estimated glomerular filtration rate (eGFR) or eGFR values less than 60 mL/min/1.73 m (2) (n = 2,956). Ultimately, 31,781 participants were included in the analysis, as depicted in Figure 1.\" Abstract, Methods: \"Participants aged 18 years and above with dietary intake and serum uric acid level information were included.\""
  },
  "exposure": {
   "definition": "Dietary Inflammatory Index (Shivappa et al. 2014) from 28 food parameters: for each, z = (intake - global mean) / global SD, converted to a percentile (0 to 1), doubled minus 1 (centered, -1 to +1), multiplied by the parameter's overall inflammatory effect score, and summed; higher = more pro-inflammatory. Categorized into tertiles of the analytic sample; headline: T2 vs T1. Whether day-1 intake or the two-day mean was used is not stated.",
   "nhanes_variables": [
    "DR1TKCAL",
    "DR1TPROT",
    "DR1TCARB",
    "DR1TFIBE",
    "DR1TTFAT",
    "DR1TSFAT",
    "DR1TMFAT",
    "DR1TPFAT",
    "DR1TCHOL",
    "DR1TBCAR",
    "DR1TNIAC",
    "DR1TFOLA",
    "DR1TMAGN",
    "DR1TIRON",
    "DR1TZINC",
    "DR1TSELE",
    "DR1TCAFF",
    "DR1TALCO",
    "DR1TP183",
    "DR1TP205",
    "DR1TP225",
    "DR1TP226",
    "DR1TP182",
    "DR1TP204",
    "DR1TVARA",
    "DR1TVB1",
    "DR1TVB2",
    "DR1TVB6",
    "DR1TVB12",
    "DR1TVC",
    "DR1TVD",
    "DR1TATOC"
   ],
   "nhanes_files": [
    "DR1TOT",
    "DR2TOT"
   ],
   "transform": "tertiles (T2 vs T1 for the headline); also per 1 unit continuous (Table 3)",
   "categories": "Methods, Statistical analysis: \"The DII score was included in the model as an independent variable, both in continuous and tertile forms (T1 (−5.28 to 0.79, n = 10,594), T2 (0.79 to 2.60, n = 10,593), T3 (2.60 to 5.79, n = 10,594)), to explore potential correlations with hyperuricemia.\"",
   "quote": "Methods, Calculation of DII: \"DII developed by Shivappa et al. was computed to measure the inflammatory potential of different dietary patterns (12). DII calculation was conducted by a standardized global database that contains daily dietary intake information of 11 regionally representative populations. Both the standard mean and standard deviation were provided for all DII food parameters from the world database.\" same section: \"The z-score for each food parameter was calculated by subtracting the standard mean from the value of consumption reported by each individual and then dividing that result by the standard deviation. These z-scores were transformed into proportions (ranging from 0 to 1) to minimize the effect of positive skewing. To obtain a symmetrical distribution centered around zero with bounds between −1 and + 1, each proportion was doubled, and then 1 was subtracted. This value was then multiplied by the corresponding inflammatory effect score for each food parameter. In this study, there were 28 parameters available in NHANES data that could be utilized to calculate DII, including energy, protein, carbohydrate, dietary fiber, total fatty acid, total saturated fatty acid, monounsaturated fatty acids (MUFA), polyunsaturated fatty acids (PUFA), cholesterol, β-carotene, niacin, folate, magnesium, iron, zinc, selenium, caffeine, alcohol, n3 polyunsaturated fatty acid, n6 polyunsaturated fatty acid, and vitamins A, B1, B2, B6, B12, C, D, and E.\""
  },
  "outcome": {
   "definition": "Hyperuricemia: serum uric acid >= 7.0 mg/dL in males, >= 6.0 mg/dL in females.",
   "nhanes_variables": [
    "LBXSUA"
   ],
   "nhanes_files": [
    "BIOPRO"
   ],
   "quote": "Methods, Serum uric acid measurement: \"Following established diagnostic standards, hyperuricemia was delineated as a serum uric acid threshold of 7.0 mg/dL or more in males and 6.0 mg/dL or more in females (16).\""
  },
  "covariates": [
   {
    "name": "age",
    "coding": "years; Table 1 mean ± SD; form in the model not stated",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "sex",
    "coding": "male, female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "BMI",
    "coding": "kg/m2; Table 1 mean ± SD; form in the model not stated",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "race/ethnicity",
    "coding": "Non-Hispanic White, Non-Hispanic Black, Mexican American, Others (Table 1)",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "educational level",
    "coding": "Less than high school, High school, More than high school (Table 1)",
    "nhanes_variables": [
     "DMDEDUC2",
     "DMDEDUC3"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "2021-2023 has DMDEDUC2 for ages 20+ only (no DMDEDUC3), so education is missing for 18-19-year-olds."
   },
   {
    "name": "smoking",
    "coding": "never (< 100 cigarettes in lifetime), former (> 100, quit), current (> 100, smoking at least every few days)",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "drinking",
    "coding": "Table 1: Never, Former, Mild, Moderate, Heavy. Current drinkers: heavy (>= 3 drinks/day women, >= 4 men, or binge drinking on 5+ days per month), moderate (>= 2 drinks/day women, >= 3 men, or binge drinking on >= 2 days per month), mild (others); never and former are not defined",
    "nhanes_variables": [
     "ALQ101",
     "ALQ110",
     "ALQ111",
     "ALQ120Q",
     "ALQ121",
     "ALQ130",
     "ALQ141Q",
     "ALQ142"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "2021-2023 has ALQ111 (ever drank), ALQ121 (past-year frequency), ALQ130 (drinks per drinking day), ALQ142 (days with 4/5+ drinks). 'Never' in the paper's cycles likely used ALQ101/ALQ110 (12+ drinks), which 2021-2023 does not ask; the categories are approximable."
   },
   {
    "name": "MET",
    "coding": "physical activity in MET per week: low (< 600), moderate (600-1,199), vigorous (>= 1,200), from minutes of 'various activities' times Compendium METs",
    "nhanes_variables": [
     "PAQ605",
     "PAQ620",
     "PAQ635",
     "PAQ650",
     "PAQ665"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": false,
    "note": "The paper's MET total appears to span work, transport and leisure (GPAQ in 2007-2018; Table 1 puts 68% of men in the vigorous group); 2021-2023 asks leisure-time activity only (PAD790Q/U, PAD800, PAD810Q/U, PAD820), so the same total cannot be built."
   },
   {
    "name": "eGFR",
    "coding": "mL/min/1.73 m2; Methods name the CKD-EPI creatinine equation but print a Cockcroft-Gault-type formula",
    "nhanes_variables": [
     "LBXSCR",
     "BMXWT"
    ],
    "nhanes_files": [
     "BIOPRO",
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "Form in the model not stated."
   },
   {
    "name": "diabetes",
    "coding": "doctor told diabetes, HbA1c > 6.5%, fasting glucose >= 7.0 mmol/L, random glucose >= 11.1 mmol/L, or OGTT 2-h glucose >= 11.1 mmol/L",
    "nhanes_variables": [
     "DIQ010",
     "LBXGH",
     "LBXGLU",
     "LBXSGL",
     "LBXGLT"
    ],
    "nhanes_files": [
     "DIQ",
     "GHB",
     "GLU",
     "BIOPRO",
     "OGTT"
    ],
    "in_2021_2023": true,
    "note": "All components exist in 2021-2023 except the OGTT (LBXGLT), which NHANES dropped after 2015-2016."
   },
   {
    "name": "hypertension",
    "coding": "BP >= 140/90 mmHg, a diagnosis of hypertension, or antihypertensive prescription",
    "nhanes_variables": [
     "BPXSY1",
     "BPXSY2",
     "BPXSY3",
     "BPXDI1",
     "BPXDI2",
     "BPXDI3",
     "BPQ020",
     "BPQ040A",
     "BPQ050A",
     "BPXOSY1",
     "BPXOSY2",
     "BPXOSY3",
     "BPXODI1",
     "BPXODI2",
     "BPXODI3",
     "BPQ150"
    ],
    "nhanes_files": [
     "BPX",
     "BPXO",
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "2021-2023: oscillometric BPXO readings and BPQ150 (taking HBP medication). Which readings were averaged is not stated."
   },
   {
    "name": "hyperlipidemia",
    "coding": "TG >= 150 mg/dL, or hypercholesterolemia (TC >= 200, LDL-C >= 130, HDL-C < 40 men / < 50 women), or lipid-lowering medication",
    "nhanes_variables": [
     "LBXTR",
     "LBXSTR",
     "LBXTC",
     "LBDLDL",
     "LBDHDD",
     "BPQ100D",
     "BPQ101D"
    ],
    "nhanes_files": [
     "TRIGLY",
     "BIOPRO",
     "TCHOL",
     "HDL",
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "Whether TG and LDL-C came from the fasting subsample (TRIGLY) or the serum biochemistry panel (LBXSTR) is not stated; 2021-2023 has both (LBXTLG, LBXSTR) and BPQ101D for medication."
   }
  ],
  "design": {
   "weights": "not stated: the paper never mentions sample weights",
   "strata_psu": "not stated",
   "quote": "Methods, Statistical analysis: \"Furthermore, a multivariable logistic regression model was utilized in both the overall population and sex-stratified subgroups to estimate odds ratios and 95% confidence intervals.\" same section: \"All statistical analyses were conducted using R (version 3.5.3) and EmpowerStats,1 with statistical significance defined as p < 0.05.\"",
   "software": "R 3.5.3 and EmpowerStats",
   "missing_data": "unstated: participants missing DII, uric acid or eGFR were excluded; handling of missing covariates is not described, and Table 3 keeps n = 31,781 in every model",
   "quote_missing": "Methods, Study population: \"Exclusion criteria were as follows: (a) participants without dietary data for DII calculation (n = 9,549), (b) participants with missing uric acid data (n = 18,995)\""
  },
  "model": {
   "family": "logistic",
   "weighted": null,
   "quote": "Methods, Statistical analysis: \"Furthermore, a multivariable logistic regression model was utilized in both the overall population and sex-stratified subgroups to estimate odds ratios and 95% confidence intervals.\" Abstract, Methods: \"Multivariable logistic regression analysis was adopted to investigate the association between DII and hyperuricemia.\""
  },
  "unstated": [
   "Whether sample weights, strata and PSUs were used at all (none are mentioned)",
   "Whether DII used day-1 intake or the mean of both recalls",
   "Which nutrients form n-3 and n-6 PUFA, and which folate variable (total folate or DFE)",
   "Whether tertile cutpoints were weighted",
   "How age, BMI, eGFR and MET enter Model 3 (continuous or grouped) and the reference categories",
   "Handling of missing covariates (Table 3 has n = 31,781 in every model)",
   "Definitions of never and former drinkers, and the items used for drinks per day and binge days",
   "Which physical activity domains count toward MET/week (work, transport, leisure)",
   "Which eGFR equation was used (CKD-EPI named, Cockcroft-Gault-type formula printed)",
   "Which blood pressure readings define hypertension, and whether TG and LDL-C came from the fasting subsample"
  ],
  "notes": "Headline rule: the abstract's first coding is T2 vs T1 (OR 1.17, 1.07-1.29, significant), so that is the headline even though the Discussion leads with T3 vs T1 (OR 1.31, 1.19-1.44, the abstract's second estimate) and the main text gives a per-unit OR of 1.06 (1.04-1.09). Weighting: the paper never mentions NHANES weights, strata or PSUs; with EmpowerStats this is most likely an unweighted analysis, but that is not stated, so model.weighted is null. eGFR: the Methods name the CKD-EPI creatinine equation but print a Cockcroft-Gault-type formula ((140 - age) x weight x 1.23 or 1.03 / creatinine in mmol/L, which gives mL/min, not per 1.73 m2, and needs creatinine in umol/L for those constants); eGFR is both an exclusion (< 60) and a covariate, so the choice changes the sample. The female hyperuricemia cut is 6.0 mg/dL (some papers use 5.7). Table 1 'Drinking' has Never and Former groups that the Methods do not define. Cycles: 2005-2018 only (70,190 = sum of the seven 2005-2006 to 2017-2018 DEMO files); the 2017-March 2020 pre-pandemic files were not used, so there is no overlap. No correction or erratum was found.",
  "adjudication": "Under the headline rule as settled, a coding with ordered categories and the lowest as reference is represented by its highest category against the reference; the extraction took the first significant contrast the abstract listed."
 },
 {
  "id": "row256",
  "rank": 188,
  "row": 256,
  "doi": "10.1038/s41598-024-66922-0",
  "pmcid": "PMC11237065",
  "title": "Association between composite dietary antioxidant index and hyperlipidemia: a cross-sectional study from NHANES (2005–2020)",
  "authors": [
   "Zhao, Minli",
   "Zhang, Danwei",
   "Zhang, Qiuping",
   "Lin, Yuan",
   "Cao, Hua"
  ],
  "year": 2024,
  "journal": "Scientific Reports",
  "table_a": {
   "predictor": "Composite dietary antioxidant index",
   "condition": "Hyperlipidemia",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "A significant negative correlation was observed between the CDAI and hyperlipidemia in the unadjusted (Odds ratio [OR] 0.97 [95% CI 0.96, 0.98]) and multi-variable adjusted (OR 0.98 [95% CI 0.97, 0.99]) models.",
   "table_location": "Table 2 'Association of composite dietary antioxidant index and hyperlipidemia', row 'Continuous', column 'Model 3' (OR (95% CI) and P)",
   "table_quote": "| Continuous | 0.97 (0.96,0.98) | < 0.001 | 0.98 (0.97,0.98) | < 0.001 | 0.98 (0,97,0.99) | < 0.001 |",
   "measure": "OR",
   "estimate": 0.98,
   "ci_low": 0.97,
   "ci_high": 0.99,
   "p_value": "< 0.001",
   "exposure_contrast": "per 1-unit increase in CDAI (continuous; CDAI is a sum of six standardized intakes)",
   "model_label": "Model 3",
   "covariates_in_this_model": [
    "age",
    "gender",
    "BMI",
    "race",
    "education levels",
    "poverty income ratio",
    "marital status",
    "alcohol consumption",
    "smoking status",
    "diabetes mellitus",
    "hypertension",
    "physical activity",
    "energy intake"
   ],
   "n_analytic": 30788,
   "n_quote": "The final analysis included 30,788 adults in the United States, among whom 25,525 (82.91%) were diagnosed with hyperlipidemia.",
   "events": 25525
  },
  "cycles": [
   "2005-2006",
   "2007-2008",
   "2009-2010",
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-2018 (inferred from the starting N, see notes)",
   "2017-March 2020 pre-pandemic (inferred from the starting N, see notes)"
  ],
  "population": {
   "age": ">=18",
   "defining": "US adults (aged 18 or older)",
   "inclusion": "NHANES 2005-2020 participants aged 18 or older with dietary recall data and the data needed to classify hyperlipidemia, not on a special diet and with plausible energy intake",
   "exclusions": [
    "age < 18 (n = 33,914)",
    "missing dietary data (n = 6261)",
    "missing hyperlipidemia data (n = 8497)",
    "special diet (n = 5914)",
    "implausible energy intake, < 500 or >= 5000 kcal/day (n = 376)"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "The data used in this study were obtained from the NHANES 2005–2020, a comprehensive cross-sectional survey that assessed the nutritional status and health of the US population21,22. ... To reduce sampling and recall bias, this study gathered data for all participants (n = 85,750) who had undergone two dietary recalls from the NHANES datasets. This study initially excluded participants (n = 33,914) who were below the age of 18. Subsequently, participants lacking dietary information (n = 6261) and those without hyperlipidemia data (n = 8497) were also excluded. Moreover, this study also excluded participants with special diet (n = 5914) and implausible energy intake (< 500 kcal/day or ≥ 5000 kcal/day) (n = 376)13. The process of participant selection is depicted in Fig. 1, and the final analysis included 30,788 eligible participants. [Figure 1 flow chart, read from the image: 'Participants form NHANES(2005-2020) (n=85750)'; 'Exclusion(n=33914) Age<18 years'; 'Participants age≥18 (N=51836)'; 'Exclusion(n=6261) Missing dietary data'; 'Remained participants (n=45575)'; 'Exclusion(n=8497) Missing hyperlipidemia data'; 'Remained participants (n=37078)'; 'Exclusion(n=6290) Special diet(n=5914) Implausible energy intake(n=376)'; 'Final analysis (n=30788)']"
  },
  "exposure": {
   "definition": "Composite dietary antioxidant index: for each of six dietary antioxidants (zinc, selenium, total carotenoids, vitamins A, C and E) the mean of the two 24-h recalls is standardized as (individual intake minus mean) / SD, and the six standardized values are summed (CDAI = sum over n = 1..6 of (Individual Intake - Mean)/SD). Intake from supplements and medications is excluded. Analyzed per 1 unit (continuous) and in quartiles. Variable names are our mapping; the paper names none.",
   "nhanes_variables": [
    "DR1TZINC",
    "DR2TZINC",
    "DR1TSELE",
    "DR2TSELE",
    "DR1TVARA",
    "DR2TVARA",
    "DR1TVC",
    "DR2TVC",
    "DR1TATOC",
    "DR2TATOC",
    "DR1TACAR",
    "DR2TACAR",
    "DR1TBCAR",
    "DR2TBCAR",
    "DR1TCRYP",
    "DR2TCRYP",
    "DR1TLYCO",
    "DR2TLYCO",
    "DR1TLZ",
    "DR2TLZ"
   ],
   "nhanes_files": [
    "DR1TOT",
    "DR2TOT"
   ],
   "transform": "none for the headline (continuous CDAI, per 1 unit); quartiles (Q1 reference) in the other models",
   "categories": null,
   "quote": "The information regarding dietary antioxidant intake was obtained through two separate 24-h dietary recall interviews23. ... Utilizing the average dietary intake data from two non-consecutive days was considered more accurate than relying solely on data from a single day16,24. Six antioxidants (zinc, selenium, total carotenoids, vitamins A, C, and E) were standardized by subtracting the mean and dividing by the standard deviation (SD). The CDAI, developed by Wright et al.25, was based on the sum of these standardized consumptions17,26. It should be noted that the calculation of dietary antioxidant intake in this study did not include antioxidants obtained from supplements, medications, or other additional sources. ... CDAI=∑n=16IndividualIntake-MeanSD."
  },
  "outcome": {
   "definition": "Hyperlipidemia (NCEP ATP III): any of TG >= 150 mg/dL, TC >= 200 mg/dL, LDL >= 130 mg/dL, HDL <= 40 mg/dL (men) or <= 50 mg/dL (women), or self-reported use of cholesterol-lowering medication. The text says 'five conditions' but lists four lab criteria plus medication use.",
   "nhanes_variables": [
    "LBXTR (TRIGLY, 2005-2020; LBXTLG in TRIGLY_L)",
    "LBXTC",
    "LBDLDL",
    "LBDHDD",
    "BPQ100D (2005-2020; BPQ101D in BPQ_L)"
   ],
   "nhanes_files": [
    "TRIGLY",
    "TCHOL",
    "HDL",
    "BPQ"
   ],
   "quote": "The hyperlipidemia status was evaluated based on the Adult Treatment Panel III (ATP 3) guidelines of the National Cholesterol Education Program (NCEP) for adults27. A diagnosis of hyperlipidemia was confirmed if any of the following five conditions were met: triglycerides (TG) ≥ 150 mg/dL, total cholesterol (TC) ≥ 200 mg/dL, low-density lipoprotein (LDL) ≥ 130 mg/dL, or high-density lipoprotein (HDL) ≤ 40 mg/dL in males and ≤ 50 mg/dL in females27. Additionally, participants who reported using cholesterol-lowering medications were also classified as having hyperlipidemia28,29."
  },
  "covariates": [
   {
    "name": "age",
    "coding": "age at screening, years (Table 1: mean (SD)); how it entered Model 3 is not stated",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "gender",
    "coding": "male or female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "BMI",
    "coding": "weight (kg) / height (m)^2; Table 1 shows mean (SD); entered presumably as continuous (not stated)",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "race",
    "coding": "non-Hispanic White, non-Hispanic Black, Mexican American, other Hispanic, other race",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Five categories match RIDRETH1."
   },
   {
    "name": "education level",
    "coding": "less than high school, high school (or equivalent), college and above",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "DMDEDUC2 is asked of ages 20+ in every cycle, so 18-19 year olds are missing (the paper used multiple imputation)."
   },
   {
    "name": "poverty income ratio (PIR)",
    "coding": "low (< 1.5), middle (1.5-3.5), high (> 3.5)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "marital status",
    "coding": "single (separated, widowed, never married, or divorced) vs married",
    "nhanes_variables": [
     "DMDMARTL (2005-2018)",
     "DMDMARTZ (2017-March 2020 and 2021-2023)"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Where 'living with partner' went is not stated; DMDMARTZ (2021-2023) codes married/living with partner together."
   },
   {
    "name": "alcohol consumption",
    "coding": "heavy: >= 3 drinks/day (women) or >= 4 drinks/day (men), or binge drinking on 5 or more days per month; moderate: >= 2 drinks/day (women) or >= 3 drinks/day (men), or binge drinking on >= 2 days per month; mild: other forms of alcohol intake",
    "nhanes_variables": [
     "ALQ130",
     "ALQ141Q/ALQ141U (2005-2016)",
     "ALQ142 (2017 onward)"
    ],
    "nhanes_files": [
     "ALQ"
    ],
    "in_2021_2023": true,
    "note": "ALQ_L has ALQ130 and ALQ142 (days with 4/5+ drinks in the past 12 months, in frequency categories), so binge days per month must be approximated from categories; how non-drinkers were coded (presumably 'mild') is not stated."
   },
   {
    "name": "smoking status",
    "coding": "never (< 100 cigarettes in life), former (> 100 cigarettes, not smoking now), current (> 100 cigarettes, smoking some days or every day)",
    "nhanes_variables": [
     "SMQ020",
     "SMQ040"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": ""
   },
   {
    "name": "diabetes mellitus",
    "coding": "diagnosed with diabetes, or using anti-diabetes drugs or insulin",
    "nhanes_variables": [
     "DIQ010",
     "DIQ070",
     "DIQ050"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "Self-report and medication only; no glucose or HbA1c criterion stated."
   },
   {
    "name": "hypertension",
    "coding": "average blood pressure exceeding 140/90 mmHg, self-reported physician-diagnosed hypertension, or currently taking antihypertensive medication",
    "nhanes_variables": [
     "BPXSY1-BPXSY4/BPXDI1-BPXDI4 (auscultatory, 2005-2018)",
     "BPXOSY1-3/BPXODI1-3 (oscillometric, 2017-March 2020 and 2021-2023)",
     "BPQ020",
     "BPQ050A (earlier cycles) / BPQ150 (2021-2023)"
    ],
    "nhanes_files": [
     "BPX/BPXO",
     "BPQ"
    ],
    "in_2021_2023": true,
    "note": "2021-2023 has only oscillometric BP (BPXO_L), a device change that keeps the construct."
   },
   {
    "name": "physical activity",
    "coding": "total MET minutes per week: weekly volume (duration x frequency) of each activity times its MET value, summed (Table 1: median (IQR) MET-min/week)",
    "nhanes_variables": [
     "PAQ605-PAD675 (GPAQ work, transport and leisure items, 2007 onward)",
     "2005-2006 PAQ/PAQIAF items"
    ],
    "nhanes_files": [
     "PAQ"
    ],
    "in_2021_2023": false,
    "note": "PAQ_L (2021-2023) has only leisure-time moderate and vigorous activity (PAD790Q/U, PAD800, PAD810Q/U, PAD820); no work or transport activity, so the paper's total MET cannot be rebuilt."
   },
   {
    "name": "energy intake",
    "coding": "average of the two 24-h dietary recalls (kcal/day)",
    "nhanes_variables": [
     "DR1TKCAL",
     "DR2TKCAL"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": ""
   }
  ],
  "design": {
   "weights": "not stated: the paper never mentions sample weights",
   "strata_psu": "not stated (not mentioned)",
   "quote": "All statistical analyses were performed using SPSS 27 and R 4.2.2 software.",
   "software": "SPSS 27 and R 4.2.2",
   "missing_data": "imputation (multiple imputation; method, number of imputations and imputed variables not stated)",
   "quote_missing": "Multiple imputation was used to handle missing values in the dataset."
  },
  "model": {
   "family": "logistic",
   "weighted": null,
   "quote": "Univariate and multivariate logistic regression models were used to examine the relationship between CDAI (including both continuous variables and quartile groups) and hyperlipidemia, with the covariates listed above. In Model 1, no covariate was adjusted. In Model 2, adjustments were made for age and gender. Model 3 included additional covariates such as BMI, race, education level, PIR, marital status, alcohol consumption, smoking status, diabetes mellitus, hypertension, physical activity, and energy intake."
  },
  "unstated": [
   "Whether sample weights, strata and PSUs were used, and which weight (the paper never mentions weighting; Table 1 counts look unweighted)",
   "Which files were used for 2017-2020: the 2017-2018 cycle files, the 2017-March 2020 pre-pandemic files, or both (the starting N fits both, see notes), and whether duplicate SEQNs were removed",
   "Which dietary variables define vitamin A (RAE or other), vitamin E (alpha-tocopherol) and total carotenoids (which carotenoids were summed)",
   "In which sample the component means and SDs for the CDAI were computed, and whether intakes were transformed or winsorized before standardizing",
   "How participants with only one reliable 24-h recall were handled (the text says two recalls; the flow chart says 'Missing dietary data')",
   "What counted as 'hyperlipidemia data': whether participants outside the fasting subsample (no TG or LDL) were classified from TC, HDL and medication alone; which LDL equation",
   "Which question defines cholesterol-lowering medication use (presumably BPQ100D, 'now taking prescribed medicine')",
   "How 'special diet' was defined (presumably DRQSDIET = yes)",
   "Multiple imputation details: method, number of imputations, imputed variables, and whether exposure or outcome were imputed",
   "How age, BMI, PIR, physical activity and energy intake entered Model 3 (continuous or categorized)",
   "Alcohol: how never and former drinkers were coded, and the reference period for binge drinking",
   "Hypertension: how many readings were averaged and whether the cutoff is > or >= 140/90 mmHg",
   "Physical activity: MET values used, and how 2005-2006 (before the GPAQ) was handled",
   "Quartile cutpoints of CDAI",
   "Where 'living with partner' was placed in marital status",
   "Whether pregnant women were excluded (no pregnancy exclusion is stated)"
  ],
  "notes": "Probable double counting of 2017-2018: Figure 1 (fetched from the PMC CDN to scratchpad/dl/supp/row256/fig1.jpg) starts from 'Participants form NHANES(2005-2020) (n=85750)'. The DEMO codebooks give 10,348 (DEMO_D), 10,149 (DEMO_E), 10,537 (DEMO_F), 9,756 (DEMO_G), 10,175 (DEMO_H), 9,971 (DEMO_I), 9,254 (DEMO_J) and 15,560 (P_DEMO) participants, which sum to exactly 85,750; without DEMO_J the sum is 76,496 and without P_DEMO it is 70,190. So the authors most likely stacked the 2017-2018 files and the 2017-March 2020 pre-pandemic files, which overlap (the 9,254 2017-2018 participants are also in P_DEMO, under new SEQNs, so the duplicates don't show as repeated SEQNs), and 2017-2018 participants may appear twice in the analytic sample. The paper does not say this; it is an inference from the counts. Reporting errors to keep in mind: the Results say 'a total of 30,388 individuals' where everything else says 30,788; Table 2 prints the Model 3 continuous lower bound as '0,97' (0.97 in the abstract and Results); Table 1 gives BMI '39.72 (± 6.75)' for the hyperlipidemia group, impossible given 29.16 overall and 26.44 without hyperlipidemia; Table 1 CDAI summaries ('− 0.61 (− 2.68, 1.94)' overall, '− 2.00 (− 2.37, 2.48)' without and '− 2.74 (− 6.91, 1.82)' with hyperlipidemia) contradict the text's 'higher CDAI levels' in hyperlipidemia and cannot all be central values of one distribution; Table 1 'Other Hispanic' non-hyperlipidemia '429 (0.82)' should be about 8.15%. Software SPSS 27 and R 4.2.2; the analysis appears unweighted. In 2021-2023, TG and LDL exist only for the fasting subsample (WTSAF2YR), and LDL has three equations (LBDLDL Friedewald, LBDLDLM, LBDLDLN). Subgroup results (age, gender, BMI, hypertension, diabetes) are in Figure 3 only and are not the headline.",
  "adjudication": null
 },
 {
  "id": "row189",
  "rank": 199,
  "row": 189,
  "doi": "10.1186/s12991-020-00315-1",
  "pmcid": "PMC7672831",
  "title": "Associations between estradiol and testosterone and depressive symptom scores of the Patient Health Questionnaire-9 in ovariectomized women: a population-based analysis of NHANES data",
  "authors": [
   "Chen, Ching-Yen",
   "Chen, Jian-Hong",
   "Ree, Shao-Chun",
   "Chang, Chun-Wei",
   "Yu, Sheng-Hsiang"
  ],
  "year": 2020,
  "journal": "Annals of General Psychiatry",
  "table_a": {
   "predictor": "Ovariectomy-reduced hormones",
   "condition": "Depression",
   "population": "US adult females"
  },
  "headline": {
   "abstract_quote": "Among ovariectomized women in the NHANES database, serum estradiol levels were significantly positively associated with PHQ-9 scores (ß = 0.014, 95% CI: 0.001, 0.028, P = 0.040), whereas serum testosterone was negatively associated with PHQ-9 scores (ß = -0.033, 95% CI: − 0.048, − 0.018, P < 0.001) after adjusting for confounders.",
   "table_location": "Table 2 'Associations between PHQ-9 scores and study variables', row 'Serum estradiol, pg/ml', columns 'Multivariate' (ß, 95% CI, P value)",
   "table_quote": "| Serum estradiol, pg/ml | − 0.003 | (− 0.011, 0.004) | 0.378 | 0.014 | (0.001, 0.028) | 0.040 |",
   "measure": "beta",
   "estimate": 0.014,
   "ci_low": 0.001,
   "ci_high": 0.028,
   "p_value": "0.040",
   "exposure_contrast": "per 1 pg/mL increase in serum estradiol (continuous, untransformed)",
   "model_label": "Multivariate (Table 2)",
   "covariates_in_this_model": [
    "Education level",
    "Serum testosterone, ng/dL",
    "Duration of estrogen use (years)",
    "CVD",
    "Chronic respiratory tract disease",
    "Arthritis"
   ],
   "n_analytic": 548,
   "n_quote": "A total of 548 women were included with a mean PHQ-9 depressive symptom score of 4.522.",
   "events": null
  },
  "cycles": [
   "2013-2014",
   "2015-2016"
  ],
  "population": {
   "age": ">20 (as written: 'women > 20 years old')",
   "defining": "women who had undergone bilateral oophorectomy (ovariectomized women)",
   "inclusion": "women older than 20 in NHANES 2013-2016 who had both ovaries removed, with serum testosterone and estradiol results",
   "exclusions": [
    "lacking results for serum testosterone or serum estradiol (count not given)"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "This cross-sectional study extracted data of women > 20 years old from the NHANES database who had undergone bilateral oophorectomy between 2013 and 2016. Those lacking results for serum testosterone or serum estradiol were excluded. ... A total of 548 women were included with a mean PHQ-9 depressive symptom score of 4.522."
  },
  "exposure": {
   "definition": "Serum estradiol (pg/mL), measured by ID-LC-MS/MS, entered as a continuous untransformed variable (per 1 pg/mL), together with serum testosterone in the same model.",
   "nhanes_variables": [
    "LBXEST"
   ],
   "nhanes_files": [
    "TST"
   ],
   "transform": "none (continuous, per 1 pg/mL)",
   "categories": null,
   "quote": "According to NHANES data documentation and laboratory analytic notes in the NHANES Laboratory Procedures Manual (https://wwwn.cdc.gov/nchs/nhanes/ContinuousNhanes/Manuals.aspx?BeginYear=2013), the concentrations of serum testosterone and estradiol were measured using an isotope dilution-liquid chromatography tandem mass spectrometry (ID-LC–MS/MS) based on the reference method of the National Institute for Standards and Technology (NIST) (details can be found at: https://wwwn.cdc.gov/Nchs/Nhanes/2013-2014/TST_H.htm)."
  },
  "outcome": {
   "definition": "PHQ-9 total score (0 to 27), sum of nine items each scored 0 to 3, analyzed as a continuous score.",
   "nhanes_variables": [
    "DPQ010",
    "DPQ020",
    "DPQ030",
    "DPQ040",
    "DPQ050",
    "DPQ060",
    "DPQ070",
    "DPQ080",
    "DPQ090"
   ],
   "nhanes_files": [
    "DPQ"
   ],
   "quote": "Briefly, depressive symptoms were assessed using the Patient Health Questionnaire-9 (PHQ-9), which has been recommended as a reliable and valid tool in community samples with good diagnostic sensitivity and specificity and good internal consistency for detecting depression symptoms of varying severity [25–27]. The PHQ-9 consists of nine items evaluating the presence of depressive symptoms during the prior 2 weeks. Each item of PHQ-9 is scored on a four-point Likert scale from 0 to 3 as follows: 0 (not at all), 1 (on several days), 2 (no more than half of the days), and 3 (nearly every day). Total scores of PHQ-9 range from 0 to 27 to measure severity of depressive symptoms, representing mild (score: 5), moderate (score: 10), moderately severe (score: 15), and severe depression (score: 20)."
  },
  "covariates": [
   {
    "name": "Education level",
    "coding": "below high school vs high school or above (reference: below high school)",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In the multivariate model."
   },
   {
    "name": "Serum testosterone, ng/dL",
    "coding": "continuous, ng/dL (ID-LC-MS/MS)",
    "nhanes_variables": [
     "LBXTST"
    ],
    "nhanes_files": [
     "TST"
    ],
    "in_2021_2023": true,
    "note": "In the multivariate model (the paper's second hormone exposure, entered together with estradiol)."
   },
   {
    "name": "Duration of estrogen use (years)",
    "coding": "years of estrogen use (derivation not stated; Table 1 mean 10.697 ± 1.108)",
    "nhanes_variables": [
     "RHQ560Q/RHQ560U",
     "RHQ576Q/RHQ576U",
     "RHQ586Q/RHQ586U",
     "RHQ602Q/RHQ602U (our guess at the source items; the paper names none)"
    ],
    "nhanes_files": [
     "RHQ"
    ],
    "in_2021_2023": false,
    "note": "In the multivariate model. RHQ_L (2021-2023) has no female hormone use questions (RHQ540-RHQ602 absent)."
   },
   {
    "name": "CVD",
    "coding": "yes/no; definition not stated",
    "nhanes_variables": [
     "MCQ160b",
     "MCQ160c",
     "MCQ160d",
     "MCQ160e",
     "MCQ160f (probable components)"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "In the multivariate model."
   },
   {
    "name": "Chronic respiratory tract disease",
    "coding": "yes/no; definition not stated",
    "nhanes_variables": [
     "MCQ010",
     "MCQ160g",
     "MCQ160k",
     "MCQ160o (probable components)"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "In the multivariate model. 2021-2023 asks asthma (MCQ010) and COPD, emphysema or chronic bronchitis in one item (MCQ160p) instead of three."
   },
   {
    "name": "Arthritis",
    "coding": "yes/no",
    "nhanes_variables": [
     "MCQ160a"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "In the multivariate model."
   },
   {
    "name": "Age",
    "coding": "<= 60 vs > 60 years",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Univariate only (not significant, not in the multivariate model)."
   },
   {
    "name": "Poverty income ratio",
    "coding": "not poor (>= 1) vs poor (< 1)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Univariate only."
   },
   {
    "name": "Married/live with partner",
    "coding": "yes/no",
    "nhanes_variables": [
     "DMDMARTL"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Univariate only; DMDMARTZ in 2021-2023."
   },
   {
    "name": "History of estrogen use",
    "coding": "yes/no (derivation not stated)",
    "nhanes_variables": [
     "RHQ540",
     "RHQ554 (probable)"
    ],
    "nhanes_files": [
     "RHQ"
    ],
    "in_2021_2023": false,
    "note": "Univariate only; also a stratifier in Table 3. Not in RHQ_L."
   },
   {
    "name": "Diabetes",
    "coding": "yes/no; definition not stated",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "Univariate only."
   },
   {
    "name": "Cancer history",
    "coding": "yes/no",
    "nhanes_variables": [
     "MCQ220"
    ],
    "nhanes_files": [
     "MCQ"
    ],
    "in_2021_2023": true,
    "note": "Univariate only."
   }
  ],
  "design": {
   "weights": "'discharge weights' (sic); which NHANES weight (e.g. WTMEC2YR, and whether divided by 2 for four years) is not stated",
   "strata_psu": "not stated",
   "quote": "All tests were applied with discharge weights to account for the NHANES sampling method, except for correlation. The NHANES database uses weights to account for the complex survey design, survey non-response, and post-stratification adjustments to match total population counts from the Census Bureau.",
   "software": "SAS 9.4",
   "missing_data": "unstated",
   "quote_missing": "Numbers may not add up to 100% due to missing value"
  },
  "model": {
   "family": "linear",
   "weighted": true,
   "quote": "Univariate and multivariate linear regression analyses were performed to assess associations between testosterone, estradiol and PHQ-9 scores. All significant variables found in univariate analysis were entered into multivariate analysis. ... All tests were applied with discharge weights to account for the NHANES sampling method, except for correlation."
  },
  "unstated": [
   "Which weight was used ('discharge weights') and whether strata and PSUs were used",
   "Whether age 20 itself was included ('women > 20 years old')",
   "How missing PHQ-9 items or scores were handled (no PHQ-9 exclusion step is stated)",
   "Which RHQ items define 'history of estrogen use' and 'duration of estrogen use (years)', how non-users and women never asked were coded, and so the multivariate model's sample size (not reported; duration is defined only for users)",
   "Definitions of CVD, chronic respiratory tract disease, arthritis, diabetes and cancer history",
   "How estradiol values below the detection limit were handled (NHANES fills them with LOD/sqrt(2))",
   "Whether hormone users at the time of the blood draw, or pregnant women, were excluded (none stated)"
  ],
  "notes": "The headline model holds both hormones and duration of estrogen use. Table 1's history of estrogen use covers only 285 of 548 women (18 'No', 267 'Yes'), consistent with the item being asked only of women who had used female hormones (our inference); if duration of estrogen use is missing for non-users and was not set to 0, the multivariate model would include at most about 267 women, but the paper never gives the model's N. Estradiol was not significant in the univariate analysis (P = 0.378) yet was entered in the multivariate model, so the stated rule ('All significant variables found in univariate analysis were entered into multivariate analysis') was not applied to the two hormones, which were forced in. For 2021-2023: duration and history of estrogen use cannot be built (RHQ_L has no hormone use items), so the covariate set must be handled by rule; chronic respiratory disease changes from three items (MCQ160g/k/o) to one (MCQ160p) plus asthma; TST_L uses the phlebotomy weight WTPH2YR. The abstract's other estimate (testosterone, beta -0.033) and the age- and estrogen-use-stratified estimates (Table 3) are not the headline. 2013-2016 only, so no 2017-2018 / 2017-March 2020 overlap. Ovariectomized women are a small group (548 in two cycles), so 2021-2023 will give a small sample.",
  "adjudication": null
 },
 {
  "id": "row315",
  "rank": 202,
  "row": 315,
  "doi": "10.3389/fnut.2024.1396470",
  "pmcid": "PMC11347418",
  "title": "Association between dietary intake of selenium and chronic kidney disease in US adults: a cross-sectional study of NHANES 2015–2018",
  "authors": [
   "Pi, Ying",
   "Liao, Xianyong",
   "Song, Xiaodan",
   "Cao, Yuyu",
   "Tang, Xiaona",
   "Lin, Guobing",
   "Zhong, Yanghong"
  ],
  "year": 2024,
  "journal": "Frontiers in Nutrition",
  "table_a": {
   "predictor": "Selenium levels",
   "condition": "Chronic kidney disease",
   "population": "US adults"
  },
  "headline": {
   "abstract_quote": "After adjusting for potential confounding factors, the fully adjusted odds ratio (OR) values for CKD according to dietary selenium intake were 1 (reference), 0.94 (95% confidence interval (CI): 0.79–1.12, p = 0.466), 0.82 (95% CI:0.68–0.98, p = 0.033), and 0.77 (95% CI:0.63–0.95, p = 0.016) for the four selenium intake levels, respectively, with P trend = 0.007.",
   "table_location": "Table 2 'Association between dietary selenium and CKD', row 'Q4' (No. 305/1528), columns 'Mode 2 OR (95% CI)' and its p-value (the most adjusted model, printed 'Mode 2')",
   "table_quote": "| Q4 | 305/1528 | 0.66 (0.56–0.78) | <0.001 | 0.77 (0.64–0.92) | 0.004 | 0.77 (0.63–0.95) | 0.016 |",
   "measure": "OR",
   "estimate": 0.77,
   "ci_low": 0.63,
   "ci_high": 0.95,
   "p_value": "0.016",
   "exposure_contrast": "Q4 (> 0.144 mg/day) vs Q1 (<= 0.072 mg/day, reference) of daily dietary selenium intake",
   "model_label": "Model 2 (column header printed 'Mode 2')",
   "covariates_in_this_model": [
    "age",
    "sex",
    "race/ethnicity",
    "marital",
    "education level",
    "body mass index (BMI)",
    "hypertension",
    "diabetes",
    "carbohydrate intake",
    "selenium supplement intake",
    "C reactive protein",
    "total cholesterol",
    "triglycerides",
    "uric acid"
   ],
   "n_analytic": 6390,
   "n_quote": "Finally, 6,390 adults (3,131 men and 3,259 women) were included in this study (Figure 1). [Methods, Study population; Table 2 first row 'No.' column: '1523/6390']",
   "events": 1523
  },
  "cycles": [
   "2015-2016",
   "2017-2018"
  ],
  "population": {
   "age": "30-80 (RIDAGEYR 30 to 80; NHANES codes ages 80 and over as 80, so 'over 80' cannot be separated)",
   "defining": "adults aged 30 to 80, both sexes (the paper's 'US adults'); no condition defines the population",
   "inclusion": "NHANES 2015-2018 participants aged 30 to 80 with demographic, laboratory (serum creatinine, urine ACR), covariate and dietary selenium data, not pregnant",
   "exclusions": [
    "Start: 'From the NHANES data from 2015–2018, a total of 19,225 participants' (Methods; Figure 1 'N = 19225')",
    "'Participants lacking basic information(N=3143)' (Methods; Figure 1: '3143 participants were excluded because they lacked basic information', leaving N = 16082)",
    "'Participants who have insufficient information for laboratory tests and other health-related survey data(N=6152)' (Methods; Figure 1 leaves N = 9930)",
    "'Participants whose age under 30 years old or over 80 years old (N=3035)' (Methods; Figure 1 leaves N = 6895)",
    "'Participants who lacking information of dietary surveys and daily selenium intake information (N=455)' (Methods; Figure 1 leaves N = 6440)",
    "'Participants who are pregnant women (N=50)' (Methods; Figure 1 leaves N = 6390)"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "From the NHANES data from 2015–2018, a total of 19,225 participants were surveyed for CKD. The inclusion criteria are as follows: (1) Participants aged 30 to 80 years old; (2) Participants who participated in the renal function blood and urine subgroup study; (3) Participants’ chronic kidney disease (CKD) status information was confirmed based on the US NHANES questionnaire data. The exclusion criteria are as follows: (1) Participants lacking basic information(N=3143); (2) Participants who have insufficient information for laboratory tests and other health-related survey data(N=6152); (3) Participants whose age under 30 years old or over 80 years old (N=3035); (4) Participants who lacking information of dietary surveys and daily selenium intake information (N=455); (5) Participants who are pregnant women (N=50) according to the USA NHANES questionnaire. Finally, 6,390 adults (3,131 men and 3,259 women) were included in this study (Figure 1). [Methods, Study population; Figure 1 (image, fetched from Europe PMC to scratchpad/dl/supp/row315/unz/) gives the same counts and the running totals 19225, 16082, 9930, 6895, 6440, 6390]"
  },
  "exposure": {
   "definition": "Daily dietary selenium intake (mg/day) from the 24-hour dietary recall (USDA AMPM; Methods say the records cover 'the respondents within 1 day', i.e. presumably the day-1 recall), categorized into quartiles of the analytic sample: Q1 <= 0.072, Q2 0.072-0.103, Q3 0.103-0.144, Q4 > 0.144 mg/day. NHANES reports selenium in mcg, so mg = mcg/1000.",
   "nhanes_variables": [
    "DR1TSELE",
    "DR2TSELE (only if a two-day mean was used; not stated)"
   ],
   "nhanes_files": [
    "DR1TOT",
    "DR2TOT"
   ],
   "transform": "mcg to mg; quartiles with Q1 as reference (headline Q4 vs Q1)",
   "categories": "Q1 (≤ 0.072 mg/day), Q2 (0.072–0.103 mg/day), Q3 (0.103–0.144 mg/day), and Q4 (> 0.144 mg/day) [Results and Table 2 note; the note calls Q 'quintile' although there are four groups]",
   "quote": "The NHANES dietary survey records refer to the data of all beverages and dietary intakes of the respondents within 1 day. From 2015 to 2018, the Automated Multiple-Pass Method (AMPM) of the United States Department of Agriculture (USDA) was used to collect the respondents’ dietary records. [Methods, Diagnosis of chronic kidney disease and dietary assessment] ... We categorized the individuals into four groups based on the selenium content in their diet, which were divided into quartiles. [Statistical analyses] ... The daily selenium intake was divided into quartiles Q1 (≤ 0.072 mg/day), Q2 (0.072–0.103 mg/day), Q3 (0.103–0.144 mg/day), and Q4 (> 0.144 mg/day) according to the quartiles. [Results]"
  },
  "outcome": {
   "definition": "CKD = eGFR < 60 mL/min/1.73 m2 by the MDRD study equation (reference 26 is Levey et al., Ann Intern Med 2006, the IDMS-traceable 4-variable MDRD: 175 x Scr^-1.154 x age^-0.203 x 0.742 if female x 1.212 if Black; the text attributes it to 'the Journal of the American Society of Nephrology in 2006'), or albuminuria = urine albumin/creatinine ratio >= 30 mg/g. Prevalence 1,523/6,390 (23.8%, unweighted).",
   "nhanes_variables": [
    "LBXSCR",
    "LBDSCRSI",
    "URDACT",
    "URXUMA",
    "URXUCR",
    "RIDAGEYR",
    "RIAGENDR",
    "RIDRETH1"
   ],
   "nhanes_files": [
    "BIOPRO",
    "ALB_CR",
    "DEMO"
   ],
   "quote": "The serum creatinine was converted to the estimated glomerular filtration rate (eGFR) according to the CKD-Modification of Diet in Renal Disease (CKD-MDRD) equation calculation formula published in the Journal of the American Society of Nephrology in 2006 (26). If the eGFR is <60 mL/ (min/1.73 m2), or albuminuria (urine albumin to creatinine ratio (ACR) ≥ 30 mg/g) is present, then CKD is diagnosed. The international standard unit for the serum creatinine level is μmol/L, and we used the serum creatinine correction formula recommended in the NHANES III database documentation for the calculation (27). [Methods, Diagnosis of chronic kidney disease and dietary assessment; XML reference 26: 'Levey AS, Coresh J, Greene T, Stevens LA, Zhang Y(L), Hendriksen S, et al. Using standardized serum creatinine values in the modification of diet in renal disease study equation for estimating glomerular filtration rate. Ann Intern Med. (2006) 145:247–54.'; reference 27: 'National Center for Health Statistics. NHANES III Laboratory data file documentation: Ages one and older. Catalog no. 76300. Hyattsville, MD: Centers for Disease Control and Prevention; (1996).']"
  },
  "covariates": [
   {
    "name": "age",
    "coding": "years (Table 1 mean ± SD); model coding not stated",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Models 1 and 2."
   },
   {
    "name": "sex",
    "coding": "male/female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Models 1 and 2."
   },
   {
    "name": "race/ethnicity",
    "coding": "Table 1: Mexican American, Non-Hispanic white, Non-Hispanic black, Others",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Models 1 and 2. Table 1's distribution (Non-Hispanic black 2,462 of 6,390, 38.5%) does not resemble the source population (22.1% non-Hispanic Black among 2015-2018 participants aged 30-80, unweighted, DEMO_I and DEMO_J), so the labels may be scrambled; see notes."
   },
   {
    "name": "marital status",
    "coding": "Married or living with partner vs Living alone",
    "nhanes_variables": [
     "DMDMARTL",
     "DMDMARTZ"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Models 1 and 2 ('marital'). DMDMARTL in 2015-2018; DEMO_L has DMDMARTZ (1 married or living with partner, 2 widowed/divorced/separated, 3 never married), enough for the two levels."
   },
   {
    "name": "education level",
    "coding": "<High School, High school or GED, >High School",
    "nhanes_variables": [
     "DMDEDUC2"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Models 1 and 2."
   },
   {
    "name": "body mass index (BMI)",
    "coding": "kg/m2 (Table 1 mean ± SD); model coding not stated",
    "nhanes_variables": [
     "BMXBMI"
    ],
    "nhanes_files": [
     "BMX"
    ],
    "in_2021_2023": true,
    "note": "In Models 1 and 2."
   },
   {
    "name": "hypertension",
    "coding": "yes/no: 'diagnosed mainly based on doctor’s notification of hypertension and systolic blood pressure ≥ 140 mmHg or diastolic blood pressure ≥ 90 mmHg'",
    "nhanes_variables": [
     "BPQ020",
     "BPXSY1-BPXSY4",
     "BPXDI1-BPXDI4",
     "BPXOSY1-BPXOSY3",
     "BPXODI1-BPXODI3"
    ],
    "nhanes_files": [
     "BPQ",
     "BPX",
     "BPXO"
    ],
    "in_2021_2023": true,
    "note": "In Models 1 and 2. 2015-2018 used auscultatory BPX (2017-2018 also has BPXO); 2021-2023 has only oscillometric BPXO_L. How self-report and measured BP combine ('and' as printed, presumably 'or') and how readings are averaged are not stated."
   },
   {
    "name": "diabetes",
    "coding": "yes/no: 'diagnosed based on doctor’s assessment and self-reported diabetes'",
    "nhanes_variables": [
     "DIQ010"
    ],
    "nhanes_files": [
     "DIQ"
    ],
    "in_2021_2023": true,
    "note": "In Models 1 and 2."
   },
   {
    "name": "carbohydrate intake",
    "coding": "g/day (Table 1 mean ± SD)",
    "nhanes_variables": [
     "DR1TCARB"
    ],
    "nhanes_files": [
     "DR1TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 2."
   },
   {
    "name": "selenium supplement intake",
    "coding": "Table 1 reports it as a count with percentage ('Selenium supplement No. (%)': 490 (10.1) non-CKD, 182 (12.0) CKD), i.e. probably yes/no; source and coding not stated",
    "nhanes_variables": [
     "DSQTSELE",
     "DS1TSELE (if the 24-hour supplement file was used)"
    ],
    "nhanes_files": [
     "DSQTOT",
     "DS1TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 2. 2021-2023 has the 30-day supplement totals DSQTOT_L DSQTSELE but no 24-hour supplement totals (DS1TOT_L is not in the 2021-2023 list)."
   },
   {
    "name": "C reactive protein",
    "coding": "'HS C-Reactive protein (mg/dL)' (Table 1 mean ± SD)",
    "nhanes_variables": [
     "LBXHSCRP"
    ],
    "nhanes_files": [
     "HSCRP"
    ],
    "in_2021_2023": true,
    "note": "In Model 2. NHANES reports hs-CRP in mg/L; Table 1 gives mg/dL."
   },
   {
    "name": "total cholesterol",
    "coding": "mmol/L (Table 1 mean ± SD)",
    "nhanes_variables": [
     "LBDTCSI"
    ],
    "nhanes_files": [
     "TCHOL"
    ],
    "in_2021_2023": true,
    "note": "In Model 2."
   },
   {
    "name": "triglycerides",
    "coding": "mmol/L (Table 1 mean ± SD)",
    "nhanes_variables": [
     "LBDSTRSI",
     "LBDTRSI"
    ],
    "nhanes_files": [
     "BIOPRO",
     "TRIGLY"
    ],
    "in_2021_2023": true,
    "note": "In Model 2. Which measurement (refrigerated serum in BIOPRO, all examinees, or the fasting subsample in TRIGLY) is not stated; N = 6,390 is too large for the fasting subsample alone."
   },
   {
    "name": "uric acid",
    "coding": "μmol/L (Table 1 mean ± SD)",
    "nhanes_variables": [
     "LBDSUASI"
    ],
    "nhanes_files": [
     "BIOPRO"
    ],
    "in_2021_2023": true,
    "note": "In Model 2."
   },
   {
    "name": "smoking",
    "coding": "smokers (more than 100 cigarettes in lifetime) vs non-smokers",
    "nhanes_variables": [
     "SMQ020"
    ],
    "nhanes_files": [
     "SMQ"
    ],
    "in_2021_2023": true,
    "note": "Listed as a covariate in Methods and in the Figure 2 (spline) adjustment and the Discussion, but NOT in the Table 2 note's Model 1 or Model 2 lists."
   },
   {
    "name": "family poverty income ratio",
    "coding": "continuous (Table 1 mean ± SD)",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "Listed in Methods and in the Figure 2 adjustment and the Discussion, but NOT in the Table 2 note's Model 1 or Model 2 lists."
   }
  ],
  "design": {
   "weights": "not stated; no sampling weight is mentioned anywhere, and Table 1 and the abstract's CKD rates by quartile are unweighted (454/1649 = 27.53%), so the analysis was probably unweighted",
   "strata_psu": "not stated (no mention of strata, PSUs or the complex design)",
   "quote": "Statistical analysis was performed using IBM SPSS software (version 25.0; IBM Corp., Armonk, NY, United States). In this study, continuous and categorical variables were expressed as mean (standard deviation, SD), and frequency variables were expressed as percentages. [Statistical analyses]",
   "software": "IBM SPSS 25.0",
   "missing_data": "complete case (participants lacking basic, laboratory, survey or dietary information were excluded)",
   "quote_missing": "The exclusion criteria are as follows: (1) Participants lacking basic information(N=3143); (2) Participants who have insufficient information for laboratory tests and other health-related survey data(N=6152); (3) Participants whose age under 30 years old or over 80 years old (N=3035); (4) Participants who lacking information of dietary surveys and daily selenium intake information (N=455)"
  },
  "model": {
   "family": "logistic",
   "weighted": null,
   "quote": "Multiple logistic regression analysis was utilized to investigate the association between dietary selenium intake and CKD, and the data are presented as odds ratios (ORs) with 95% confidence intervals (CIs). [Statistical analyses]"
  },
  "unstated": [
   "whether sampling weights, strata and PSUs were used (nothing is said; the analysis looks unweighted)",
   "whether the exposure is the day-1 recall only (Methods: 'within 1 day') or a two-day mean (abstract: 'average selenium intakes')",
   "whether selenium from supplements is part of the exposure (it enters Model 2 separately as 'selenium supplement intake')",
   "how the quartile cutpoints were computed (sample quartiles of the 6,390, weighted or not) and which quartile takes values equal to 0.072, 0.103 and 0.144 mg/day (the printed ranges share endpoints; Q1 is '≤ 0.072' and Q4 '> 0.144')",
   "the exact eGFR equation: the text names a CKD-MDRD formula 'published in the Journal of the American Society of Nephrology in 2006' while reference 26 is the Levey 2006 Annals paper (IDMS-traceable MDRD, coefficient 175); and what 'serum creatinine correction formula recommended in the NHANES III database documentation' was applied to 2015-2018 creatinine",
   "how race enters the MDRD equation (which code counts as Black)",
   "what inclusion criterion (3), CKD status 'confirmed based on the US NHANES questionnaire data', means for the outcome (the definition given is laboratory-based)",
   "how participants coded 80 (ages 80 and over) were treated, given the exclusion of those 'over 80 years old'",
   "model coding of age, BMI, CRP, cholesterol, triglycerides, uric acid and carbohydrate (continuous or categorized) and the reference categories of race, marital status and education",
   "which triglyceride measurement was used (refrigerated serum LBXSTR or the fasting subsample LBXTR)",
   "how 'selenium supplement intake' was measured and coded (yes/no, amount; 24-hour or 30-day supplement file)",
   "how hypertension combines self-report and measured blood pressure, and which readings were averaged",
   "whether smoking and the poverty income ratio were in Model 2 (the Table 2 note omits them; Figure 2 and the Discussion include them)",
   "the SD used for the per-SD estimate and the unit of Table 2's unlabeled first row"
  ],
  "notes": "Headline choice: under the study lead's rule for ordered categories, the abstract's quartile coding is represented by Q4 vs Q1 (OR 0.77, 0.63-0.95, p = 0.016), not by the first-listed Q2 vs Q1 (0.94, 0.79-1.12, p = 0.466). Suchak et al.'s label 'Selenium levels' is the paper's DIETARY selenium intake (24-hour recall), not blood selenium (PBCD LBXBSE). Exposure check: the abstract's CKD rates by quartile are unweighted counts from Table 2 (454/1649 = 27.53%, 407/1621 = 25.11%, 357/1592 = 22.42%, 305/1528 = 19.96%), consistent with an unweighted analysis in SPSS. Internal inconsistencies: (1) the Statistical analyses text says Model 1 adjusted for 'age, sex, and race', while the Table 2 note's Model 1 also has marital status, education, BMI, hypertension and diabetes; (2) the Table 2 note says 'Q, quintile' for four groups; (3) the per-SD row prints Model 1 OR 0.93 (0.81-1.05) with p = 0.025 although the CI includes 1, and Model 2 0.93 (0.83-1.09) with p = 0.059 (an asymmetric CI); (4) the abstract's '7.7% for every additional 0.1 mg' cannot be traced to Table 2: the unlabeled first row (Model 2 OR 0.23 per presumably 1 mg/day) would imply 0.23^0.1 = 0.86 per 0.1 mg, a 14% decrease, and the per-SD OR 0.93 is a 7% decrease per SD (SD about 0.07 mg/day in Table 1); (5) Table 1's race distribution (Non-Hispanic black 38.5%, Non-Hispanic white 26.9%, Mexican American 14.0%, Others 20.6% of 6,390) does not resemble 2015-2018 participants aged 30-80 (unweighted RIDRETH1 from DEMO_I and DEMO_J: Mexican American 14.8%, Other Hispanic 11.4%, non-Hispanic White 34.4%, non-Hispanic Black 22.1%, other 17.2%), so the race labels may be mixed up, which matters for both the race covariate and the MDRD race coefficient; (6) the Table 1 smoking rows print two p-values (0.004 and <0.001). The text's 'renal function blood and urine subgroup study' is not an NHANES subsample: serum creatinine (BIOPRO) and urine albumin/creatinine (ALB_CR) are measured on all eligible examinees. NHANES 2015-2018 creatinine needs no calibration correction (the NHANES III correction cited as reference 27 applies to older surveys). Figures 1-3 were read from the images fetched with the Europe PMC supplementaryFiles service (scratchpad/dl/supp/row315/unz/); Figure 3's subgroup ORs (e.g. '< 65years ... 0.64 (0.15–2.77)') appear to be per mg/day of selenium. No 2017-2018 / 2017-March 2020 overlap. No erratum.",
  "adjudication": null
 },
 {
  "id": "row066",
  "rank": 206,
  "row": 66,
  "doi": "10.1371/journal.pone.0300566",
  "pmcid": "PMC11146693",
  "title": "The association of caffeine intake and prevalence of obesity among children and adolescents: A cross-sectional survey from NHANES 2011–2020 March",
  "authors": [
   "Liu, Zi Rui",
   "Cui, Kai"
  ],
  "year": 2024,
  "journal": "PLOS ONE",
  "table_a": {
   "predictor": "Caffeine intake",
   "condition": "Obesity",
   "population": "US children and adolescents"
  },
  "headline": {
   "abstract_quote": null,
   "table_location": "Table 2 'Association between caffeine intake and obesity', row 'Quartile 4' (printed cut-off '≥708.05'), column 'Model 3' (OR (95% CI), P value)",
   "table_quote": "| Quartile 4 | 1.0008 (0.9996, 1.0019) 0.1765, <0.0001 | 1.0009 (0.9997, 1.0022) 0.1344, <0.0001 | 1.0020 (1.0005, 1.0036) 0.0086 <0.0001 |",
   "measure": "OR",
   "estimate": 1.002,
   "ci_low": 1.0005,
   "ci_high": 1.0036,
   "p_value": "0.0086 (printed '0.0086 <0.0001'; the second number in each cell is unexplained)",
   "exposure_contrast": "Quartile 4 (printed '≥708.05') vs Quartile 1 (printed '≤236.25', reference) of energy-adjusted (residual method) two-day mean caffeine intake. The printed cut-offs have no units and do not match Table 1's caffeine means by quartile (0.00, 1.79, 9.86 and 70.09 mg/day).",
   "model_label": "Model 3",
   "covariates_in_this_model": [
    "gender",
    "age",
    "race",
    "the ratio of family income to poverty",
    "total intake (from dietary and supplements)",
    "energy",
    "protein",
    "total sugar",
    "total fat",
    "cholesterol",
    "vitamin B6",
    "vitamin C",
    "calcium",
    "phosphorus",
    "zinc",
    "copper",
    "sodium",
    "phosphorus",
    "selenium"
   ],
   "n_analytic": 10001,
   "n_quote": "This study had 16,225 participants between the ages of 2 and 19 years. Then, the participants who had no data on caffeine and body mass index (BMIC) were also excluded. The final study population of this study was 10001. [Methods, Study population; Abstract: 'A total of 10,001 classified children and adolescents were included in this analysis.'; Table 1's age-group counts sum to 10,001]",
   "events": null
  },
  "cycles": [
   "2011-2012",
   "2013-2014",
   "2015-2016",
   "2017-March 2020 (pre-pandemic P files)"
  ],
  "population": {
   "age": "2-19",
   "defining": "children and adolescents aged 2 to 19, both sexes",
   "inclusion": "participants aged 2-19 in NHANES 2011-March 2020 with caffeine (two 24-hour recalls) and BMI-category data",
   "exclusions": [
    "Start (Figure 1): 'Paticipants from NHANES 2011-2020 March (N=45,932)' (the four DEMO files hold 45,462 participants: 9,756 + 10,175 + 9,971 + 15,560)",
    "'Paticipants without The Age of 2-19 (N=29,707)' (Figure 1), leaving 'N=16,225'",
    "'Paticipants without The Caffeine (N=6,032)' (Figure 1), leaving 'N=10,193'",
    "'Paticipants without The Obesity (N=192)' (Figure 1), leaving 'N=10,001'"
   ],
   "exclusions_not_in_2021_2023": [],
   "quote": "This study analyzed the data from 2011–2020 March. This study had 16,225 participants between the ages of 2 and 19 years. Then, the participants who had no data on caffeine and body mass index (BMIC) were also excluded. The final study population of this study was 10001. The flowchart of the participants is shown in Fig 1. [Methods, Study population; Fig 1 (image, fetched with the Europe PMC supplementaryFiles service into scratchpad/dl/supp/row066/unz/pone.0300566.g001.jpg; 'Paticipants' as printed): 'Paticipants from NHANES 2011-2020 March (N=45,932)'; 'Paticipants without The Age of 2-19 (N=29,707)'; 'N=16,225'; 'Paticipants without The Caffeine (N=6,032)'; 'N=10,193'; 'Paticipants without The Obesity (N=192)'; 'N=10,001']"
  },
  "exposure": {
   "definition": "Caffeine intake (mg/day): the mean of the two 24-hour dietary recalls (total nutrient intake files), adjusted for total energy intake with the residual method, then categorized into quartiles (Q1 reference). Table 1's mean caffeine by quartile is 0.00, 1.79, 9.86 and 70.09 mg/day (overall 20.43 ± 18.86).",
   "nhanes_variables": [
    "DR1TCAFF",
    "DR2TCAFF",
    "DR1TKCAL",
    "DR2TKCAL"
   ],
   "nhanes_files": [
    "DR1TOT",
    "DR2TOT"
   ],
   "transform": "two-day mean; energy adjustment by the residual method; quartiles with Q1 as reference (headline Q4 vs Q1); also a 'Per Quarter' quartile score",
   "categories": "| Quartile 1 | 1.0000 (reference) | 1.0000 (reference) | 1.0000 (reference) | ... | ≤236.25 | ... | 236.25–472.5 | ... | 472.5–708.05 | ... | ≥708.05 | [Table 2 rows; Table 1 header prints the same four ranges under Quartile 1 to 4; no units, and they do not match caffeine in mg/day]",
   "quote": "All NHANES participants were eligible for two 24-hour dietary recall interviews. The first dietary recall interview was conducted in person at the Mobile Examination Center (MEC), and the second interview was conducted by telephone 3–10 days later. Daily totals of nutrients/food components for all foods were calculated for NHANES data collection using the USDA Dietary Study Food and Nutrition Database, including approximately 50 coffee beverages, 30 teas, and caffeinated beverages. Therefore, the analysis used the average caffeine intake from two total nutrient intake recalls. [Methods, Exposure variable] ... Caffeine was adjusted for total energy intake with a residual model. [Methods, Statistical analysis] ... The average caffeine intake of the overall participants was 20.43 ± 18.86 mg/day, and the averages of caffeine intake for quartiles 1–4 were 0.00 ± 0.03, 1.79 ± 0.95, 9.86 ± 4.90 and 70.09 ± 69.55, respectively. [Results]"
  },
  "outcome": {
   "definition": "Obesity from the NHANES variable 'BMI Category - Children/Youth' (BMDBMIC: CDC BMI-for-age percentile categories 1 underweight, 2 normal, 3 overweight, 4 obese). Which categories form the outcome is not stated; the Methods say the ORs are 'for overweight and obese individuals', and the weighted prevalence 34.87% matches overweight or obese (BMDBMIC 3 or 4) rather than obese alone (see notes).",
   "nhanes_variables": [
    "BMDBMIC",
    "BMXBMI"
   ],
   "nhanes_files": [
    "BMX"
   ],
   "quote": "BMI was calculated as body weight in kilograms divided by height in meters squared. The BMI of children and adolescents is based on age and gender and is often called the age-based BMI (BMI-for-age). The Centers for Disease Control (CDC) compiles these BMI values into age-based BMI growth charts for boys and girls to obtain a percentile ranking. [Methods, Outcome variable] ... This study used the data from the NHANES 2011–2020 March ‘Body Measures–BMI Category [em dash in the original] Children/Youth’ for children and adolescents as the research sample. [Methods, Outcome variable] ... Weighted logistic regression was used to calculate odds ratios (ORs) and 95% confidence intervals (CIs) for overweight and obese individuals for each quartile of caffeine intake [Methods, Statistical analysis]"
  },
  "covariates": [
   {
    "name": "gender",
    "coding": "male/female",
    "nhanes_variables": [
     "RIAGENDR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Models 2 and 3."
   },
   {
    "name": "age",
    "coding": "continuous ('age (continuous)')",
    "nhanes_variables": [
     "RIDAGEYR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Models 2 and 3."
   },
   {
    "name": "race",
    "coding": "Table 1: Mexican American, Other Hispanic, Non-Hispanic White, Non-Hispanic Black, Other Race (including multi-racial); 'characteristics with three or more categories as indicator variables'",
    "nhanes_variables": [
     "RIDRETH1"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Models 2 and 3."
   },
   {
    "name": "ratio of family income to poverty",
    "coding": "continuous (Table 1 mean (SD))",
    "nhanes_variables": [
     "INDFMPIR"
    ],
    "nhanes_files": [
     "DEMO"
    ],
    "in_2021_2023": true,
    "note": "In Model 3."
   },
   {
    "name": "total intake (from dietary and supplements)",
    "coding": "not explained",
    "nhanes_variables": [
     "DS1TOT/DS2TOT totals (if 24-hour supplement intakes were meant; not stated)"
    ],
    "nhanes_files": [
     "DS1TOT",
     "DS2TOT"
    ],
    "in_2021_2023": false,
    "note": "In Model 3 per the Table 2 note and Results; its meaning is unclear. If it means nutrient totals including 24-hour supplement intakes, 2021-2023 has only the 30-day supplement totals (DSQTOT_L), not DS1TOT/DS2TOT."
   },
   {
    "name": "energy",
    "coding": "kcal, two-day mean",
    "nhanes_variables": [
     "DR1TKCAL",
     "DR2TKCAL"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3."
   },
   {
    "name": "protein",
    "coding": "g, two-day mean, energy-adjusted by residuals",
    "nhanes_variables": [
     "DR1TPROT",
     "DR2TPROT"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3. The Table 1 note says 'Caffeine and dietary confounders (e.g., minerals and vitamins) were adjusted for total energy intake with a residual model'; which nutrients were residual-adjusted is not listed."
   },
   {
    "name": "total sugar",
    "coding": "g, two-day mean",
    "nhanes_variables": [
     "DR1TSUGR",
     "DR2TSUGR"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3."
   },
   {
    "name": "total fat",
    "coding": "g, two-day mean",
    "nhanes_variables": [
     "DR1TTFAT",
     "DR2TTFAT"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3."
   },
   {
    "name": "cholesterol",
    "coding": "mg, two-day mean",
    "nhanes_variables": [
     "DR1TCHOL",
     "DR2TCHOL"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3."
   },
   {
    "name": "vitamin B6",
    "coding": "mg, two-day mean",
    "nhanes_variables": [
     "DR1TVB6",
     "DR2TVB6"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3."
   },
   {
    "name": "vitamin C",
    "coding": "mg, two-day mean",
    "nhanes_variables": [
     "DR1TVC",
     "DR2TVC"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3."
   },
   {
    "name": "calcium",
    "coding": "mg, two-day mean",
    "nhanes_variables": [
     "DR1TCALC",
     "DR2TCALC"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3."
   },
   {
    "name": "phosphorus",
    "coding": "mg, two-day mean",
    "nhanes_variables": [
     "DR1TPHOS",
     "DR2TPHOS"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3 (listed twice in the Table 2 note)."
   },
   {
    "name": "zinc",
    "coding": "mg, two-day mean",
    "nhanes_variables": [
     "DR1TZINC",
     "DR2TZINC"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3."
   },
   {
    "name": "copper",
    "coding": "mg, two-day mean",
    "nhanes_variables": [
     "DR1TCOPP",
     "DR2TCOPP"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3."
   },
   {
    "name": "sodium",
    "coding": "mg, two-day mean",
    "nhanes_variables": [
     "DR1TSODI",
     "DR2TSODI"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3."
   },
   {
    "name": "selenium",
    "coding": "two-day mean (Table 1 labels it mg; NHANES reports mcg)",
    "nhanes_variables": [
     "DR1TSELE",
     "DR2TSELE"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In Model 3 per the Table 2 note; not in the Methods' Model 3 list."
   },
   {
    "name": "potassium",
    "coding": "mg, two-day mean",
    "nhanes_variables": [
     "DR1TPOTA",
     "DR2TPOTA"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In the Methods' Model 3 list, but not in the Table 2 note."
   },
   {
    "name": "magnesium",
    "coding": "mg, two-day mean",
    "nhanes_variables": [
     "DR1TMAGN",
     "DR2TMAGN"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In the Methods' Model 3 list, but not in the Table 2 note."
   },
   {
    "name": "vitamin D",
    "coding": "two-day mean",
    "nhanes_variables": [
     "DR1TVD",
     "DR2TVD"
    ],
    "nhanes_files": [
     "DR1TOT",
     "DR2TOT"
    ],
    "in_2021_2023": true,
    "note": "In the Methods' Model 3 list ('vitamins B6, C, and D'), but not in the Table 2 note."
   }
  ],
  "design": {
   "weights": "'combined dietary sample weights' (variables not named; with two-day mean intakes presumably the two-day dietary weights WTDR2D for 2011-2016 and WTDR2DPP for 2017-March 2020); how the 2-year and 3.2-year weights were combined is not stated",
   "strata_psu": "the complex sampling design was accounted for in SUDAAN; strata and PSU variables not named",
   "quote": "All statistical analyses were conducted according to CDC guidelines (https://wwwn.cdc.gov/nchs/nhanes/tutorials/default.aspx). Sample weight was considered and assigned to each participant [15]. [Methods, Statistical analysis] ... All statistical analyses were conducted in SAS version 9.4 (SAS Institute, Cary, North Carolina) or SAS-callable SUDAAN (RTI International, Raleigh, North Carolina) with combined dietary sample weights for nonresponse and the complex sampling design. [Methods, Statistical analysis]",
   "software": "SAS 9.4 and SAS-callable SUDAAN",
   "missing_data": "imputation (as stated: 8 multiple imputations by fully conditional specification), but the paragraph is about un-geocoded addresses and appears copied from another study; participants missing caffeine or BMI category were excluded",
   "quote_missing": "After excluding samples of 2–19 years old missing data about caffeine intake or obesity, addresses for 78.23% (35,931 out of 45,932) of the participants could not be geocoded and contributed to missing data in cross-sectional analyses. As such, 8 multiple imputations using fully conditional specifications addressed potential biases arising from item nonresponse. [Methods, Missing covariables]"
  },
  "model": {
   "family": "logistic",
   "weighted": true,
   "quote": "Weighted logistic regression was used to calculate odds ratios (ORs) and 95% confidence intervals (CIs) for overweight and obese individuals for each quartile of caffeine intake, and this study calculated three different logistic regression models. [Methods, Statistical analysis]"
  },
  "unstated": [
   "whether the outcome is obesity (BMDBMIC = 4, BMI-for-age >= 95th percentile) or overweight or obesity (BMDBMIC 3 or 4); the reference group (all others, or normal weight only)",
   "what the quartile cut-offs (≤236.25, 236.25–472.5, 472.5–708.05, ≥708.05) are measured in, and how quartiles were formed (weighted or not; on raw or energy-adjusted caffeine), given Q1's mean caffeine of 0.00 mg/day and unequal quartile sizes",
   "how the residual energy adjustment was done (regression of caffeine on energy in which sample; whether the mean was added back) and whether nutrient covariates were also residual-adjusted",
   "whether participants needed both recalls (the two-day mean) or day 1 only",
   "which weight (two-day dietary weight presumably) and how the 2011-2016 two-year weights and the 2017-March 2020 weights were combined",
   "what 'total intake (from dietary and supplements)' is",
   "which covariates Model 3 really has (the Methods' list adds potassium, magnesium and vitamin D and omits selenium; the Table 2 note lists phosphorus twice)",
   "which variables were multiply imputed, with what model, and how estimates were combined",
   "race coding in the model (five groups as in Table 1, or white vs nonwhite as in the subgroup analysis)",
   "what the second number printed in each Table 2 cell ('<0.0001', '<0.027', '<0.03') means",
   "how the starting count 45,932 in Figure 1 was obtained (the four DEMO files hold 45,462)"
  ],
  "notes": "Headline choice: the abstract gives no whole-population estimate with a 95% CI, so Table 2 decides; it has no continuous-caffeine estimate, so Q4 vs Q1 in Model 3 is the headline, OR 1.0020 (1.0005, 1.0036), P 0.0086. Alternative if 'Per Quarter' (the quartile score) is taken as the continuous coding: OR 1.0022 (1.0009, 1.0036), P 0.0015. WARNING for the coding check: Table 2's whole-population ORs are almost certainly not quartile contrasts. Table 1's weighted obesity rates are 33.80% (Q1) and 40.77% (Q4), a crude OR near 1.35, but Table 2's crude Q4 vs Q1 OR is 1.0008; the paper's own subgroup Q4 vs Q1 ORs in Table 4 are 1.5961 (boys) and 1.4418 (girls), which cannot pool to a whole-population 1.0020. Values like 1.0020 look like per-unit (per mg or per cut-off unit) ORs mislabeled as quartile contrasts, so a faithful re-run of Q4 vs Q1 should not be expected to reproduce 1.0020. Other inconsistencies: the text's 'per-quartile ... 0.05% increased prevalence' (Model 3) and '0.03% decreased prevalence' (Model 1) match no printed OR (Table 2's per-quarter ORs are 1.0022 and 1.0009, both above 1); the text gives an inflection point of 479.15 while Table 3 prints 26.99, with a CI '(1.0009, 1.0013)' that excludes its own estimate 1.0022; typos such as '(0.9999, 10023)', '(0.9924, 1597)' and '(0.9688, 13849)'; the abstract's subgroup sentence garbles which OR belongs to which group ('per-quartile 1.3497 (1.2014, 1.5163)' is the boys' Model 3 per-quarter OR in Table 4, and '1.3181 (1.0613, 1.6370)' is the nonwhite Q4 OR); 'body mass index-children and adolescents (BMIC)' is listed among covariates although it is the outcome; Table 1's 'Other Race' Q2 count 214 should be 428 for the race counts to sum to the age-group total of the quartile. Outcome check on the paper's own cycles (local DEMO, BMX and DR1TOT/DR2TOT files for 2011-2012, 2013-2014, 2015-2016 and 2017-March 2020): among 2-19 year olds with both recalls and a BMI category, the pooled weighted prevalence of BMDBMIC 3 or 4 is 35.37% (two-day weights scaled 2/9.2 and 3.2/9.2; 34.99% with day-1 weights among day-1 completers) and of BMDBMIC 4 alone 19.32%, so the paper's 34.87% 'rate of obesity' is overweight or obesity. Sample check: 16,225 aged 2-19 (matches); 10,960 have both recalls and BMDBMIC, 10,106 also have PIR, 12,896 have day-1 caffeine and BMDBMIC; none equals Figure 1's 10,193 with caffeine or the final 10,001. The two-day mean caffeine in the 10,960 averages 19.98 mg/day unweighted (SD 41.4) against the paper's 20.43 ± 18.86. No 2017-2018 / 2017-March 2020 overlap. Figures and table images were fetched with the Europe PMC supplementaryFiles service (scratchpad/dl/supp/row066/unz/); the Table 2 image matches the text conversion. Our check scripts are in scratchpad/scripts/x_r315_198_066_244/ (row066_counts.R, row066_counts2.R).",
  "adjudication": "The coding check (section 6.3) found that Table 2's cells are not contrasts between quartiles: each is the per-mg odds ratio of caffeine among the participants in one quartile of caffeine intake, from unweighted models, labeled as that quartile against quartile 1 (its crude and minimally adjusted cells are reproduced to three decimals). The headline is unchanged (Table 2's Model 3 cell for quartile 4); the re-implementation computes it as the paper did, and the quartile contrast the text describes is a variant."
 }
]