{"id":"W4399772788","doi":"10.2196/60428","title":"Peer Review of “Performance Drift in Machine Learning Models for Cardiac Surgery Risk Prediction: Retrospective Analysis”","year":2024,"lang":"en","type":"article","venue":"JMIRx Med","topic":"Cardiac, Anesthesia and Surgical Outcomes","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Retrospective cohort study; Medicine; Artificial intelligence; Computer science; Machine learning; Internal medicine","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06473716,0.001013596,0.002515379,0.00540847,0.003433443,0.007797462,0.003573145,0.005077813,0.07487048],"category_scores_gemma":[0.4073098,0.0007587101,0.002567338,0.004217384,0.002647903,0.004340659,0.004100913,0.00432597,0.04896701],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002883703,"about_ca_system_score_gemma":0.01285274,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003328495,"about_ca_topic_score_gemma":0.005128493,"domain_scores_codex":[0.9361157,0.02488939,0.007917809,0.003727039,0.02569637,0.001653696],"domain_scores_gemma":[0.3926927,0.0564791,0.0177611,0.0294092,0.4944139,0.009243939],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001086123,0.0000160683,0.0007686724,0.0008731349,0.00009686426,0.00009827701,0.0001277505,0.00008008463,0.0002414627,0.0005623918,0.9687582,0.02826853],"study_design_scores_gemma":[0.0001873385,0.00008838929,0.004207335,0.00287112,0.000165466,0.0004212327,0.0004101174,0.001961856,0.001547489,0.00357771,0.9844472,0.0001146496],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"editorial","genre_gemma":"empirical","genre_scores_codex":[0.00658628,0.01057663,0.02330591,0.3266858,0.5853113,0.003075786,0.009146927,0.003769074,0.03154228],"genre_scores_gemma":[0.1382407,0.02811165,0.05473826,0.1373949,0.3850988,0.005862372,0.02678212,0.008933217,0.214838],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9352629,"threshold_uncertainty_score":0.3423669,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02528289339510343,"score_gpt":0.2890196440600437,"score_spread":0.2637367506649402,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}