{"id":"W4412806735","doi":"10.1186/s12874-025-02631-0","title":"Using a large language model (ChatGPT) to assess risk of bias in randomized controlled trials of medical interventions: protocol for a pilot study of interrater agreement with human reviewers","year":2025,"lang":"en","type":"article","venue":"BMC Medical Research Methodology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"Norwegian Institute of Public Health","keywords":"Inter-rater reliability; Psychological intervention; Randomized controlled trial; Systematic review; Protocol (science); MEDLINE; Medicine; Clinical trial; Psychology; Medical physics; Family medicine; Physical therapy; Rating scale; Nursing; Alternative medicine; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3105822,0.007108857,0.007371102,0.008176475,0.005104705,0.004843885,0.005260194,0.008471794,0.05330199],"category_scores_gemma":[0.4244253,0.005981152,0.01205591,0.008252203,0.00697935,0.007306986,0.006473519,0.0117041,0.01626089],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01250082,"about_ca_system_score_gemma":0.03694225,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002830311,"about_ca_topic_score_gemma":0.00649187,"domain_scores_codex":[0.6840594,0.2239194,0.05856463,0.01269777,0.01578455,0.004974222],"domain_scores_gemma":[0.5142837,0.2559751,0.06330341,0.07851917,0.08067188,0.007246806],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.1842701,0.009862248,0.007192956,0.2324544,0.008198351,0.002039657,0.02025876,0.01843023,0.01172074,0.0300106,0.1766368,0.2989253],"study_design_scores_gemma":[0.3315027,0.03187759,0.02083384,0.09170981,0.005974621,0.001125647,0.005395067,0.0389812,0.0161413,0.06691469,0.3870388,0.002504758],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"protocol","genre_gemma":"protocol","genre_scores_codex":[0.000688322,0.0001067181,0.007074327,0.0002128296,0.0001820472,0.9900284,0.001051859,0.0002678305,0.0003875714],"genre_scores_gemma":[0.0003319764,0.00001952308,0.005764198,0.00003234706,0.000006776899,0.9937047,0.00007267214,0.000009549443,0.00005817259],"genre_candidate":"protocol","genre_consensus":"protocol","teacher_disagreement_score":0.6894178,"threshold_uncertainty_score":0.8501749,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9568881869498969,"score_gpt":0.7730588343050259,"score_spread":0.183829352644871,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}