{"id":"W4391873644","doi":"10.1093/jcag/gwad061.105","title":"A105 PILOT STUDY ON THE ACCURACY OF CHATGPT IN ARTICLE SCREENING FOR SYSTEMATIC REVIEWS IN GASTROENTEROLOGY","year":2024,"lang":"en","type":"article","venue":"Journal of the Canadian Association of Gastroenterology","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"The Scarborough Hospital; Queen's University; St. Michael's Hospital","funders":"","keywords":"Systematic review; Medicine; Medical physics; Internal medicine; Gastroenterology; MEDLINE; Chemistry; Biochemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.5689362,0.001531926,0.003187579,0.007303155,0.002100594,0.005592619,0.002812024,0.003736737,0.005891496],"category_scores_gemma":[0.878897,0.00191584,0.009069335,0.009691936,0.003426579,0.01119716,0.007344394,0.003034867,0.001332658],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005699816,"about_ca_system_score_gemma":0.009226698,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005029335,"about_ca_topic_score_gemma":0.007667873,"domain_scores_codex":[0.3266521,0.5161854,0.1050636,0.01469637,0.034691,0.002711597],"domain_scores_gemma":[0.03508918,0.8685918,0.03317015,0.03161445,0.03073286,0.0008015743],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.03320337,0.002740717,0.3446495,0.1092456,0.01508755,0.001442975,0.05136069,0.007231155,0.003968032,0.009382796,0.02177984,0.3999078],"study_design_scores_gemma":[0.03062104,0.04879065,0.4375268,0.08315581,0.05994065,0.006933115,0.01947834,0.1635988,0.02124237,0.02852496,0.09823791,0.001949582],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.582345,0.01521222,0.2229539,0.01955219,0.002337896,0.1243392,0.01053281,0.003433914,0.01929294],"genre_scores_gemma":[0.7409436,0.001218705,0.1882478,0.003356676,0.0002980174,0.06350912,0.001548518,0.0002794589,0.0005980858],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4310638,"threshold_uncertainty_score":0.5315785,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4165001958900607,"score_gpt":0.4455626395568268,"score_spread":0.02906244366676619,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}