{"id":"W4412806735","doi":"10.1186/s12874-025-02631-0","title":"Using a large language model (ChatGPT) to assess risk of bias in randomized controlled trials of medical interventions: protocol for a pilot study of interrater agreement with human reviewers","year":2025,"lang":"en","type":"article","venue":"BMC Medical Research Methodology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"Norwegian Institute of Public Health","keywords":"Inter-rater reliability; Psychological intervention; Randomized controlled trial; Systematic review; Protocol (science); MEDLINE; Medicine; Clinical trial; Psychology; Medical physics; Family medicine; Physical therapy; Rating scale; Nursing; Alternative medicine; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3280509,0.0001578956,0.004959648,0.001265975,0.00005965292,0.000006346855,0.000363147,0.0001947114,0.0006527906],"category_scores_gemma":[0.587743,0.00008977206,0.0004906176,0.0006965634,0.0004192535,0.00003008356,0.0001885738,0.0006164501,6.553689e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001209794,"about_ca_system_score_gemma":0.003332389,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006568152,"about_ca_topic_score_gemma":0.01328867,"domain_scores_codex":[0.9149309,0.07413664,0.006890054,0.0005507331,0.002864315,0.0006273508],"domain_scores_gemma":[0.8437575,0.1503021,0.001859484,0.0008613216,0.002682519,0.0005371014],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"randomized_trial","study_design_gemma":"randomized_trial","study_design_scores_codex":[0.9005108,0.01498645,0.0143112,0.02069057,0.001777925,0.00001877768,0.0163517,0.0001732699,0.002760484,0.0049364,0.0005842252,0.02289815],"study_design_scores_gemma":[0.6029182,0.04343224,0.0006252311,0.07062028,0.002169534,0.00001090747,0.1007497,0.1521267,0.01553632,0.01136149,0.00008271348,0.0003666961],"study_design_candidate":"randomized_trial","study_design_consensus":"randomized_trial","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3686716,0.0001997264,0.4697412,0.0007428211,0.00007877324,0.1605023,0.000006387023,0.000005002111,0.00005211689],"genre_scores_gemma":[0.6799417,0.00008973158,0.05179708,0.000163233,0.0001120822,0.2676779,0.000008916209,0.00001918514,0.000190176],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4179441,"threshold_uncertainty_score":0.9929126,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9568881869498969,"score_gpt":0.7730588343050259,"score_spread":0.183829352644871,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}