{"id":"W4400119997","doi":"10.2196/51614","title":"Detecting Algorithmic Errors and Patient Harms for AI-Enabled Medical Devices in Randomized Controlled Trials: Protocol for a Systematic Review","year":2024,"lang":"en","type":"review","venue":"JMIR Research Protocols","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Birmingham Biomedical Research Centre; Department of Health and Social Care; National Institute for Health and Care Research","keywords":"Protocol (science); Randomized controlled trial; Computer science; Medical physics; Medicine; Systematic review; MEDLINE; Alternative medicine; Data mining; Pathology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.09077216,0.006725075,0.0244219,0.01205774,0.003585048,0.00940942,0.004645379,0.009083771,0.06133963],"category_scores_gemma":[0.1642188,0.004853047,0.02600818,0.01283429,0.006167378,0.009727401,0.005819634,0.007976539,0.008612677],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0171705,"about_ca_system_score_gemma":0.03713313,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003936837,"about_ca_topic_score_gemma":0.00950352,"domain_scores_codex":[0.9205506,0.0380061,0.02370251,0.005510308,0.009558427,0.002672018],"domain_scores_gemma":[0.9129808,0.04063397,0.02200319,0.00708583,0.01520611,0.002090089],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","study_design_scores_codex":[0.005988785,0.0002067088,0.0006835922,0.948878,0.008661577,0.0002734143,0.0007745242,0.0008604671,0.0006576691,0.003594188,0.01411268,0.01530848],"study_design_scores_gemma":[0.06235542,0.00272116,0.004661583,0.7585304,0.03449002,0.0004625706,0.001436279,0.00269905,0.002024105,0.01628112,0.1137923,0.0005459439],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","genre_codex":"protocol","genre_gemma":"protocol","genre_scores_codex":[0.0003666756,0.003257881,0.001685878,0.0004234299,0.0004413032,0.9899422,0.003213197,0.0001426972,0.0005267723],"genre_scores_gemma":[0.000481206,0.0008193372,0.00213681,0.0002241967,0.00002289322,0.995899,0.0002225728,0.00001178972,0.0001821686],"genre_candidate":"protocol","genre_consensus":"protocol","teacher_disagreement_score":0.9092278,"threshold_uncertainty_score":0.4800548,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7044931414052633,"score_gpt":0.7231471353513194,"score_spread":0.01865399394605605,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}