{"id":"W4399207674","doi":"10.2196/56537","title":"Evaluating Literature Reviews Conducted by Humans Versus ChatGPT: Comparative Study","year":2024,"lang":"en","type":"article","venue":"JMIR AI","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":42,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ottawa Public Health; University of Ottawa; Ottawa Hospital; Queen's University; Canadian Medical Protective Association","funders":"","keywords":"Preprint; Computer science; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2717398,0.001368196,0.00510691,0.03056249,0.001664261,0.007655391,0.002672063,0.00219036,0.004208188],"category_scores_gemma":[0.6600004,0.001181433,0.004325212,0.0225284,0.002801949,0.008530358,0.006139343,0.001017847,0.0009693857],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.008371261,"about_ca_system_score_gemma":0.01080611,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00152999,"about_ca_topic_score_gemma":0.004408862,"domain_scores_codex":[0.5764719,0.2672432,0.09169594,0.01159976,0.05101505,0.001974177],"domain_scores_gemma":[0.1450186,0.6584218,0.09263445,0.01368284,0.08644775,0.003794533],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01459204,0.001849215,0.08645887,0.3104185,0.015836,0.001670121,0.1063084,0.001318997,0.003656352,0.00218548,0.00772696,0.447979],"study_design_scores_gemma":[0.01048701,0.03609189,0.3633178,0.2834513,0.04610786,0.006859279,0.1398821,0.007147506,0.01066567,0.006508641,0.08799723,0.001483629],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.680958,0.1992193,0.02779307,0.004303267,0.001635667,0.05998091,0.008723745,0.0006248186,0.01676118],"genre_scores_gemma":[0.830635,0.04588488,0.06306921,0.00238875,0.0008388978,0.05243974,0.003289928,0.0002160358,0.001237646],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7282603,"threshold_uncertainty_score":0.8980746,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5298481920861103,"score_gpt":0.6187488228927178,"score_spread":0.08890063080660748,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}