{"id":"W4414112257","doi":"10.1002/cesm.70048","title":"Using a Large Language Model (ChatGPT‐4o) to Assess the Risk of Bias in Randomized Controlled Trials of Medical Interventions: Interrater Agreement With Human Reviewers","year":2025,"lang":"en","type":"article","venue":"Cochrane Evidence Synthesis and Methods","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"Norwegian Institute of Public Health","keywords":"Inter-rater reliability; Agreement; Randomized controlled trial; Model validation; Measure (data warehouse); Calibration","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.7619197,0.007370433,0.01964782,0.02351708,0.005820741,0.009930045,0.008500907,0.008243685,0.0187987],"category_scores_gemma":[0.8882137,0.006235834,0.02877616,0.01571547,0.009498737,0.01418926,0.01631079,0.008507168,0.005145483],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01285737,"about_ca_system_score_gemma":0.02923201,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003941874,"about_ca_topic_score_gemma":0.008409406,"domain_scores_codex":[0.1067533,0.6603938,0.1749592,0.02463094,0.03133505,0.001927691],"domain_scores_gemma":[0.04613531,0.7663724,0.06828541,0.0547078,0.06311962,0.001379381],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"systematic_review","study_design_gemma":"observational","study_design_scores_codex":[0.03453084,0.0008431542,0.02860078,0.4588304,0.07092933,0.0009230326,0.03903034,0.01376084,0.003669414,0.02734542,0.0524018,0.2691348],"study_design_scores_gemma":[0.08738573,0.01019054,0.04010155,0.2405607,0.09705469,0.001927976,0.009705169,0.1383158,0.01629209,0.1655956,0.1874374,0.005432922],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"protocol","genre_gemma":"empirical","genre_scores_codex":[0.02250313,0.01907771,0.4138569,0.006371999,0.005594089,0.5045269,0.01377909,0.004495549,0.009794598],"genre_scores_gemma":[0.05709551,0.001038298,0.2464436,0.001223659,0.000314312,0.6912498,0.001458107,0.0004457431,0.0007309859],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2380803,"threshold_uncertainty_score":0.2935954,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5139431960532691,"score_gpt":0.6271990136593266,"score_spread":0.1132558176060575,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}