{"id":"W4414112257","doi":"10.1002/cesm.70048","title":"Using a Large Language Model (ChatGPT‐4o) to Assess the Risk of Bias in Randomized Controlled Trials of Medical Interventions: Interrater Agreement With Human Reviewers","year":2025,"lang":"en","type":"article","venue":"Cochrane Evidence Synthesis and Methods","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"Norwegian Institute of Public Health","keywords":"Inter-rater reliability; Agreement; Randomized controlled trial; Model validation; Measure (data warehouse); Calibration","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.09627951,0.0001570908,0.00325074,0.0003452853,0.00007448524,0.00002018819,0.0001719278,0.0001065972,0.000243606],"category_scores_gemma":[0.1580775,0.00007693006,0.0005953229,0.0003411063,0.0001850274,0.00007338555,0.00006414482,0.0002078361,4.761609e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005054226,"about_ca_system_score_gemma":0.0003439967,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001464159,"about_ca_topic_score_gemma":0.0003699602,"domain_scores_codex":[0.9877259,0.008795921,0.002561895,0.0002565798,0.0004684053,0.0001913203],"domain_scores_gemma":[0.9598582,0.03751583,0.001347434,0.0006018565,0.0005308379,0.0001458467],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"systematic_review","study_design_scores_codex":[0.1719638,0.001820577,0.02189498,0.02281536,0.004073557,0.00002084256,0.03997339,0.00065605,0.04301243,0.007937344,0.000435637,0.685396],"study_design_scores_gemma":[0.04693123,0.0008943421,0.001769082,0.4116149,0.01684579,0.00002502807,0.05587371,0.1879879,0.2726413,0.004677044,0.00009248589,0.0006472041],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.533897,0.04079063,0.4175783,0.004406045,0.000186421,0.003020365,0.000004915934,0.000007355941,0.0001089056],"genre_scores_gemma":[0.9568168,0.01086654,0.03132135,0.0003169342,0.00004693592,0.0004916049,0.000001061835,0.000008593995,0.0001302138],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6847488,"threshold_uncertainty_score":0.9305704,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5139431960532691,"score_gpt":0.6271990136593266,"score_spread":0.1132558176060575,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}