{"id":"W6958960482","doi":"10.6084/m9.figshare.c.7959736","title":"Using a large language model (ChatGPT) to assess risk of bias in randomized controlled trials of medical interventions: protocol for a pilot study of interrater agreement with human reviewers","year":2025,"lang":"en","type":"other","venue":"Figshare","topic":"Plant pathogens and resistance mechanisms","field":"Agricultural and Biological Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Inter-rater reliability; Protocol (science); Psychological intervention; Randomized controlled trial; Systematic review; MEDLINE; Meta-analysis; Clinical trial","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3467451,0.006859313,0.007771284,0.008309552,0.005251568,0.005108572,0.005055854,0.008184808,0.05195412],"category_scores_gemma":[0.4854593,0.00621909,0.01225604,0.008342841,0.007360243,0.007621805,0.006310721,0.01156856,0.01639496],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01224657,"about_ca_system_score_gemma":0.03436245,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002472697,"about_ca_topic_score_gemma":0.005554856,"domain_scores_codex":[0.645754,0.2479333,0.06911009,0.01357108,0.01835724,0.005274341],"domain_scores_gemma":[0.4524781,0.2909864,0.06933989,0.08762946,0.09190964,0.007656556],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.1961729,0.01040061,0.007440032,0.2495847,0.008671843,0.002123194,0.02307052,0.01927898,0.01208048,0.03017687,0.1506281,0.2903718],"study_design_scores_gemma":[0.3675791,0.03633482,0.0202087,0.09765606,0.006212618,0.001166094,0.005949905,0.04484247,0.01848354,0.06687858,0.3320337,0.002654562],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"protocol","genre_gemma":"protocol","genre_scores_codex":[0.0007646016,0.00009784894,0.00697588,0.0001920051,0.0002017322,0.9903145,0.0008581718,0.000240513,0.0003546005],"genre_scores_gemma":[0.0003767313,0.00001766332,0.005069899,0.00003070106,0.00000771145,0.9943807,0.00005822105,0.000009135958,0.00004931252],"genre_candidate":"protocol","genre_consensus":"protocol","teacher_disagreement_score":0.6532549,"threshold_uncertainty_score":0.8055795,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3481904797074832,"score_gpt":0.4344037178980377,"score_spread":0.08621323819055449,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}