{"id":"W6958211007","doi":"10.6084/m9.figshare.29755807.v1","title":"Additional file 2 of Using a large language model (ChatGPT) to assess risk of bias in randomized controlled trials of medical interventions: protocol for a pilot study of interrater agreement with human reviewers","year":2025,"lang":"en","type":"article","venue":"Figshare","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Protocol (science); Inter-rater reliability; Randomized controlled trial; Calibration; Measure (data warehouse); Data collection","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.04745063,0.002559709,0.004273036,0.004667237,0.002284057,0.00377056,0.002030039,0.003444972,0.8040789],"category_scores_gemma":[0.3148936,0.003388669,0.006080781,0.004684545,0.001699064,0.003392175,0.002758608,0.003486087,0.07946248],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003452336,"about_ca_system_score_gemma":0.007321985,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002394342,"about_ca_topic_score_gemma":0.008450217,"domain_scores_codex":[0.9669926,0.02099701,0.00693709,0.002325867,0.001909706,0.0008377437],"domain_scores_gemma":[0.5469527,0.4021093,0.019029,0.01701274,0.01288459,0.002011697],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.02400703,0.00127734,0.003010625,0.1052608,0.002752734,0.0002011123,0.001163074,0.003894184,0.0004825326,0.01181509,0.7731833,0.07295198],"study_design_scores_gemma":[0.3904147,0.004418572,0.020458,0.0529304,0.00554238,0.0006509474,0.001290764,0.02664482,0.002328443,0.09330779,0.4009211,0.001092142],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"protocol","genre_scores_codex":[0.002992158,0.0004657707,0.02764571,0.00142311,0.0007445692,0.1883046,0.7650114,0.004964412,0.008448222],"genre_scores_gemma":[0.01147128,0.0001713829,0.06662462,0.00119734,0.0001727005,0.881895,0.02836558,0.001293277,0.008808896],"genre_candidate":"protocol","genre_consensus":null,"teacher_disagreement_score":0.9525493,"threshold_uncertainty_score":0.2794576,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6326181384729683,"score_gpt":0.5831750434507937,"score_spread":0.04944309502217459,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}