{"id":"W4408139986","doi":"10.1093/ije/dyaf013","title":"Protocol for improving equity in quantitative big data cleaning: lessons from longitudinal analysis of electronic health records from underrepresented and marginalized communities","year":2025,"lang":"en","type":"article","venue":"International Journal of Epidemiology","topic":"Ethics in Clinical Research","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute on Aging; National Institutes of Health","keywords":"Health equity; Protocol (science); Equity (law); Underrepresented Minority; Health records; Big data; Longitudinal data; Business; Environmental health; Data science; Medicine; Gerontology; Political science; Sociology; Computer science; Economic growth; Medical education; Public health; Data mining; Demography; Health care; Nursing; Economics; Alternative medicine","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","open_science"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4966466,0.002437209,0.002708928,0.006537691,0.009443928,0.007195106,0.006945765,0.009740822,0.01889147],"category_scores_gemma":[0.6358802,0.003461948,0.005362968,0.009773224,0.009586449,0.007604843,0.01080135,0.0145772,0.007380452],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.008886718,"about_ca_system_score_gemma":0.06008006,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005508658,"about_ca_topic_score_gemma":0.005327994,"domain_scores_codex":[0.4834611,0.4258966,0.04699421,0.0132523,0.02662479,0.003770894],"domain_scores_gemma":[0.3505227,0.2562712,0.02717507,0.2214932,0.1376459,0.006891991],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.006640761,0.001614688,0.009257186,0.01707877,0.001595462,0.001577905,0.03667101,0.007146493,0.004702926,0.2662288,0.2460228,0.4014632],"study_design_scores_gemma":[0.01030842,0.002907518,0.02056068,0.02807819,0.001049291,0.000947292,0.009358495,0.03013169,0.0171607,0.2873613,0.5908635,0.001272817],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"protocol","genre_scores_codex":[0.005776543,0.0005187624,0.4952405,0.01405969,0.003217102,0.4644532,0.006230279,0.001889651,0.008614331],"genre_scores_gemma":[0.008160874,0.0001889982,0.3040611,0.002050422,0.0002092204,0.6831906,0.0008792706,0.0001429991,0.001116451],"genre_candidate":"protocol","genre_consensus":null,"teacher_disagreement_score":0.9930542,"threshold_uncertainty_score":0.6207244,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.8596062526297745,"score_gpt":0.7186874689373219,"score_spread":0.1409187836924526,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}