{"id":"W4386152215","doi":"10.1093/postmj/qgad069","title":"Limitations of large language models in medical applications","year":2023,"lang":"en","type":"letter","venue":"Postgraduate Medical Journal","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"University of British Columbia; Faculty of Medicine, University of British Columbia","keywords":"Medicine; Data science; Natural language processing; Bioinformatics; Medical physics; Computer science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["research_integrity"],"consensus_categories":[],"category_scores_codex":[0.001946511,0.0002311768,0.000636882,0.0007048312,0.0001158713,0.00002240686,0.0004119492,0.001225781,0.0006567331],"category_scores_gemma":[0.004113054,0.000193302,0.000242656,0.0007238368,0.0001938248,0.00008632238,0.00006940705,0.005343558,0.0002712681],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001879729,"about_ca_system_score_gemma":0.002996419,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000431565,"about_ca_topic_score_gemma":0.0003942747,"domain_scores_codex":[0.9947584,0.0002386707,0.00148253,0.0003084267,0.002582856,0.0006291661],"domain_scores_gemma":[0.9966917,0.001523448,0.0003568434,0.0003518431,0.0004505644,0.0006256037],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00004061956,0.0003902029,0.0006757454,0.0004850033,0.0001065395,0.003477784,0.002968861,0.000006283026,0.000008360737,0.0003148351,0.7761592,0.2153666],"study_design_scores_gemma":[0.001131703,0.0007696355,0.001698605,0.008135243,0.000543752,0.006975741,0.007305522,0.02100729,0.0003114861,0.03810268,0.9130121,0.00100627],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.005809038,0.001001813,0.004511754,0.9858778,0.0009346924,0.0005314064,0.00004504653,0.00006209751,0.001226344],"genre_scores_gemma":[0.1303713,0.01686827,0.0007721976,0.8264205,0.02092269,0.000320797,0.001643747,0.0001964504,0.002484009],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.2143603,"threshold_uncertainty_score":0.9969512,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3033259786544968,"score_gpt":0.4571410420054396,"score_spread":0.1538150633509427,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}