{"id":"W4392691776","doi":"10.1186/s12911-024-02459-6","title":"Assessing the research landscape and clinical utility of large language models: a scoping review","year":2024,"lang":"en","type":"review","venue":"BMC Medical Informatics and Decision Making","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":167,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canada Research Chairs; University of Toronto; University of Calgary; University of New Brunswick","funders":"","keywords":"CINAHL; MEDLINE; Health care; Medicine; Socioeconomic status; Political science; Environmental health; Population","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1718375,0.002301046,0.008593314,0.03542912,0.002120293,0.01103298,0.005124259,0.00607639,0.006227976],"category_scores_gemma":[0.5482943,0.002852559,0.009953997,0.02450619,0.005300195,0.01355393,0.006477953,0.00469463,0.001038149],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01172921,"about_ca_system_score_gemma":0.04309619,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01122525,"about_ca_topic_score_gemma":0.02468225,"domain_scores_codex":[0.8631908,0.07186358,0.04198496,0.005928912,0.01579751,0.001234285],"domain_scores_gemma":[0.3119388,0.6074064,0.03392983,0.009120462,0.03639677,0.001207718],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0002265477,0.00005404937,0.002430344,0.7518304,0.007039542,0.000214237,0.001412222,0.0007621812,0.0002260212,0.004350704,0.00519191,0.2262618],"study_design_scores_gemma":[0.00004991621,0.00008972125,0.001000401,0.9715098,0.008602347,0.0001539304,0.0006418822,0.0003575545,0.0001274283,0.002617395,0.01480679,0.00004284365],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.001117371,0.9870391,0.003216455,0.005160247,0.0005220579,0.001334286,0.0004338985,0.00003759756,0.001138964],"genre_scores_gemma":[0.03498712,0.9394551,0.01495771,0.003727278,0.0005491539,0.005261468,0.0007529237,0.00006519533,0.000244047],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.8281624,"threshold_uncertainty_score":0.9087746,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6937181054312056,"score_gpt":0.6891676052751003,"score_spread":0.004550500156105253,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}