{"id":"W4327862079","doi":"10.1101/2023.03.14.23287202","title":"Characterizing the limitations of using diagnosis codes in the context of machine learning for healthcare","year":2023,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Institute for Clinical Evaluative Sciences; SickKids Foundation; Hospital for Sick Children","funders":"","keywords":"Medicine; Context (archaeology); Confidence interval; Medical diagnosis; Pediatrics; Odds ratio; Gold standard (test); Concordance; Kappa; Internal medicine; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2091266,0.001162026,0.001275795,0.004245313,0.001463091,0.008234139,0.003020379,0.002241843,0.001388456],"category_scores_gemma":[0.6547476,0.0008023886,0.00142931,0.005515145,0.003472916,0.007080661,0.004258971,0.004597299,0.0005825665],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002945988,"about_ca_system_score_gemma":0.004417709,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009120115,"about_ca_topic_score_gemma":0.008070958,"domain_scores_codex":[0.6516013,0.2974032,0.01700932,0.01104157,0.02158718,0.001357518],"domain_scores_gemma":[0.1649779,0.7605585,0.02354551,0.02303123,0.02648558,0.001401202],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001264228,0.0002524827,0.6906855,0.002127633,0.001974124,0.0002378905,0.003674296,0.03108644,0.0006981987,0.02141264,0.01235266,0.234234],"study_design_scores_gemma":[0.0002626996,0.001040473,0.2571768,0.006779002,0.0008114768,0.0009934608,0.006046799,0.5009661,0.004298677,0.1899243,0.03134455,0.0003555962],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5480855,0.02409496,0.3192057,0.06811021,0.002343336,0.001131329,0.005944738,0.0006598326,0.0304245],"genre_scores_gemma":[0.9324551,0.001176081,0.0614787,0.002302643,0.0009083467,0.0003892386,0.0008607492,0.00010474,0.0003244599],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7908734,"threshold_uncertainty_score":0.9752877,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2051099917473286,"score_gpt":0.3675477546853069,"score_spread":0.1624377629379783,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}