{"id":"W4398175922","doi":"10.2196/51187","title":"The Use of Generative AI for Scientific Literature Searches for Systematic Reviews: ChatGPT and Microsoft Bing AI Performance Evaluation","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":45,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Data science; Information retrieval; World Wide Web; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004500464,0.00011932,0.000314185,0.0001350402,0.0003524575,0.0003850079,0.00009030246,0.0001453546,0.000008636219],"category_scores_gemma":[0.004115119,0.00006605775,0.00009141009,0.0003528777,0.0002456649,0.0005188962,0.00002427555,0.0002579471,0.000008641194],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000851592,"about_ca_system_score_gemma":0.0006718987,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000002409851,"about_ca_topic_score_gemma":0.000006491575,"domain_scores_codex":[0.9979421,0.00008541477,0.001040958,0.0001166995,0.0005836971,0.0002310539],"domain_scores_gemma":[0.9969509,0.001547808,0.0001669278,0.0002314589,0.0009566685,0.000146282],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001518198,0.00009295299,0.0002178812,0.3222093,0.0001245527,6.164813e-7,0.1249402,0.00002461006,0.0004044053,0.002382388,0.05559937,0.4938518],"study_design_scores_gemma":[0.0000973533,0.0002980505,0.00002154447,0.03625681,0.000116571,0.00002902816,0.002660084,0.9065404,0.00308193,0.0004143947,0.05038642,0.00009739606],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8817834,0.02712014,0.03878512,0.02326606,0.002801038,0.0260419,0.00007664253,0.00009494323,0.00003073274],"genre_scores_gemma":[0.9555642,0.006955301,0.01553763,0.007643403,0.001514609,0.0099679,0.0007367599,0.00006206517,0.00201815],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9065158,"threshold_uncertainty_score":0.4926479,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3173253048606453,"score_gpt":0.4986945925217383,"score_spread":0.181369287661093,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}