{"id":"W4388591773","doi":"10.1001/jamacardio.2023.4859","title":"Natural Language Processing for Adjudication of Heart Failure in a Multicenter Clinical Trial","year":2023,"lang":"en","type":"letter","venue":"JAMA Cardiology","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Heart, Lung, and Blood Institute","keywords":"Adjudication; Medicine; Heart failure; Concordance; Clinical trial; Medical record; Gold standard (test); Cohort; Intensive care medicine; Emergency medicine; Pediatrics; Internal medicine","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.6299868,0.001916322,0.004793076,0.004231583,0.002184804,0.005849863,0.004190527,0.003458426,0.007334777],"category_scores_gemma":[0.7121432,0.001818562,0.007231195,0.004463239,0.005259517,0.005503962,0.005350755,0.00505869,0.002071221],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002722536,"about_ca_system_score_gemma":0.01030349,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008310165,"about_ca_topic_score_gemma":0.001401131,"domain_scores_codex":[0.1434277,0.7831157,0.04424493,0.01468854,0.01368622,0.000836962],"domain_scores_gemma":[0.1358904,0.670931,0.09095584,0.08173177,0.0179934,0.002497657],"domain_codex":"methods","domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.1032505,0.004015435,0.1939941,0.05149685,0.07875043,0.000977912,0.006238417,0.03387637,0.00477416,0.03111456,0.08071723,0.4107941],"study_design_scores_gemma":[0.1099781,0.03394189,0.1905489,0.02344799,0.0205867,0.001626322,0.001654008,0.3662836,0.009638256,0.1154246,0.1253332,0.001536454],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09494375,0.009600507,0.6848547,0.0166282,0.005026093,0.1610022,0.01151765,0.004140316,0.01228662],"genre_scores_gemma":[0.4148662,0.0006271657,0.3851381,0.005369356,0.001645556,0.1869632,0.004402023,0.0003559804,0.0006323622],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6299868,"threshold_uncertainty_score":0.4562922,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04726143691355651,"score_gpt":0.3963823846668722,"score_spread":0.3491209477533157,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}