{"id":"W4399772732","doi":"10.2196/60280","title":"Peer Review of “Performance Drift in Machine Learning Models for Cardiac Surgery Risk Prediction: Retrospective Analysis”","year":2024,"lang":"en","type":"article","venue":"JMIRx Med","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Medicine; Computer science; Cardiology; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005093067,0.0001699308,0.0007015793,0.0005076948,0.00009697235,0.00005306994,0.000349176,0.00008755005,0.00002145437],"category_scores_gemma":[0.001337618,0.000152277,0.0003919232,0.002478503,0.00002715877,0.0004425556,0.0001097942,0.0006646851,0.000005971832],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001465521,"about_ca_system_score_gemma":0.000163754,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002160298,"about_ca_topic_score_gemma":0.00002927801,"domain_scores_codex":[0.9973,0.0004150594,0.0006362266,0.0005531063,0.0007893136,0.0003062539],"domain_scores_gemma":[0.9979368,0.0006778014,0.0002156537,0.0005275475,0.0005614454,0.00008077517],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002131292,0.00004202909,0.8864877,0.009366462,0.0005070515,0.00000970252,0.002020488,0.03907138,0.000007201241,0.004546801,0.01189636,0.04602356],"study_design_scores_gemma":[0.00005612101,0.00006484762,0.05213534,0.001490388,0.0001125299,0.000001638438,0.000006503217,0.9163346,0.00002157811,0.0003335862,0.02931139,0.0001314642],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04271655,0.1352878,0.7671583,0.03916004,0.004279518,0.003874066,0.0005046782,0.001679357,0.005339646],"genre_scores_gemma":[0.9805555,0.01043989,0.004582786,0.0001131373,0.0001657021,0.000287885,0.0001252438,0.00002673486,0.003703068],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.937839,"threshold_uncertainty_score":0.6209675,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03087608632671638,"score_gpt":0.3063412251667841,"score_spread":0.2754651388400677,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}