{"id":"W4386152215","doi":"10.1093/postmj/qgad069","title":"Limitations of large language models in medical applications","year":2023,"lang":"en","type":"letter","venue":"Postgraduate Medical Journal","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"University of British Columbia; Faculty of Medicine, University of British Columbia","keywords":"Medicine; Data science; Natural language processing; Bioinformatics; Medical physics; Computer science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.056511,0.0009996311,0.001936474,0.00261363,0.001742153,0.008576273,0.004490891,0.004693311,0.01060644],"category_scores_gemma":[0.3435676,0.001616466,0.00200651,0.003155824,0.003942322,0.01554662,0.004987033,0.01135694,0.006274247],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005647178,"about_ca_system_score_gemma":0.00753876,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02213811,"about_ca_topic_score_gemma":0.02014985,"domain_scores_codex":[0.9428516,0.04173766,0.003624994,0.003146556,0.007942324,0.000696942],"domain_scores_gemma":[0.4927408,0.4657661,0.004221779,0.01504159,0.01992043,0.00230928],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001089637,0.0002782508,0.01239394,0.002664144,0.0008724239,0.001362927,0.001742788,0.07888763,0.0009221903,0.1515797,0.3007567,0.4474496],"study_design_scores_gemma":[0.0002400979,0.0001568662,0.002163285,0.001582643,0.0002283128,0.00170923,0.0007088038,0.344168,0.0009491217,0.4878146,0.1600765,0.0002023465],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.01684872,0.03838225,0.3731355,0.52198,0.005909964,0.0004889017,0.005025789,0.004107165,0.03412164],"genre_scores_gemma":[0.5492093,0.03752621,0.2593976,0.1045726,0.02251875,0.001823018,0.007004711,0.002778996,0.01516885],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.056511,"threshold_uncertainty_score":0.2988623,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3033259786544968,"score_gpt":0.4571410420054396,"score_spread":0.1538150633509427,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}