{"id":"W4388894207","doi":"10.1101/2023.11.19.23298727","title":"ChatGPT for assessing risk of bias of randomized trials using the RoB 2.0 tool: A methods study","year":2023,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"Impact; McMaster University; University of Toronto","funders":"","keywords":"Randomized controlled trial; Computer science; Risk analysis (engineering); Medicine; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.502461,0.005951729,0.01491977,0.0269741,0.00230773,0.008232013,0.005331236,0.005958365,0.02404883],"category_scores_gemma":[0.7995621,0.005580769,0.03709953,0.02104036,0.005344057,0.007362287,0.009102753,0.00579849,0.002736429],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007497956,"about_ca_system_score_gemma":0.01889664,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002397771,"about_ca_topic_score_gemma":0.004416952,"domain_scores_codex":[0.246781,0.5718535,0.1253459,0.01762611,0.03652182,0.001871777],"domain_scores_gemma":[0.08901643,0.7938649,0.0552926,0.03454721,0.02619502,0.001083862],"domain_codex":"methods","domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.01449661,0.0004857732,0.02490961,0.5618593,0.1502167,0.000933344,0.006462311,0.006586024,0.001518852,0.01891349,0.03128719,0.1823308],"study_design_scores_gemma":[0.06599366,0.01023945,0.04278004,0.3271938,0.2511848,0.004933523,0.002712064,0.08461566,0.007960914,0.07094513,0.1283902,0.003050832],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01699727,0.04194381,0.5362647,0.004179752,0.002940618,0.3669125,0.01891291,0.006362879,0.005485498],"genre_scores_gemma":[0.05491256,0.002786728,0.3998675,0.001009007,0.0002958032,0.5384712,0.001341649,0.0006022481,0.000713358],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.497539,"threshold_uncertainty_score":0.6135542,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7182047399308633,"score_gpt":0.6179040126192602,"score_spread":0.1003007273116031,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}