{"id":"W4388894207","doi":"10.1101/2023.11.19.23298727","title":"ChatGPT for assessing risk of bias of randomized trials using the RoB 2.0 tool: A methods study","year":2023,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"Impact; McMaster University; University of Toronto","funders":"","keywords":"Randomized controlled trial; Computer science; Risk analysis (engineering); Medicine; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.07637895,0.000238827,0.003099336,0.0003025793,0.0001316959,0.00003864618,0.0002310311,0.0002569648,0.00002829],"category_scores_gemma":[0.1013572,0.0001423009,0.00102705,0.0003245138,0.0001746558,0.00004051143,0.0001596485,0.0005066418,0.000001454368],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006937087,"about_ca_system_score_gemma":0.0008271168,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004517387,"about_ca_topic_score_gemma":0.00006480955,"domain_scores_codex":[0.9878557,0.00785592,0.003187431,0.0004273033,0.0004220823,0.0002515742],"domain_scores_gemma":[0.962066,0.03268057,0.003253873,0.0009461663,0.0009857403,0.00006765036],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.07324555,0.002694681,0.3401664,0.009285335,0.00925691,0.00001023099,0.1251156,0.00799132,0.01507714,0.0004206709,0.0004114238,0.4163247],"study_design_scores_gemma":[0.05118975,0.001707103,0.04115525,0.01140479,0.04937154,0.00001926327,0.1443766,0.2540261,0.3231204,0.1213378,0.000616918,0.001674432],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8291721,0.0007111136,0.1601871,0.0006947546,0.002669508,0.006492021,0.00002122209,0.00003560224,0.00001659234],"genre_scores_gemma":[0.9501875,0.0004128457,0.04774228,0.00003660206,0.0008036792,0.0006458408,0.00002022414,0.00005379594,0.00009725168],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4146503,"threshold_uncertainty_score":0.9510623,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7182047399308633,"score_gpt":0.6179040126192602,"score_spread":0.1003007273116031,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}