{"id":"W4392193048","doi":"10.1038/s41591-024-02855-5","title":"Adapted large language models can outperform medical experts in clinical text summarization","year":2024,"lang":"en","type":"article","venue":"Nature Medicine","topic":"Topic Modeling","field":"Computer Science","cited_by":664,"is_retracted":false,"has_abstract":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Institute of Biomedical Imaging and Bioengineering; Foundation for the National Institutes of Health; National Institute of Arthritis and Musculoskeletal and Skin Diseases; Agency for Healthcare Research and Quality; National Institutes of Health; National Heart, Lung, and Blood Institute","keywords":"Automatic summarization; Computer science; Natural language processing; Artificial intelligence; Information retrieval; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005629766,0.001541467,0.001116136,0.002364814,0.0005290118,0.001794644,0.001003833,0.001835313,0.002782171],"category_scores_gemma":[0.02221637,0.0005136384,0.001443735,0.001298014,0.0003036874,0.002192709,0.001107723,0.002158841,0.003270875],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005169972,"about_ca_system_score_gemma":0.001357205,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003054273,"about_ca_topic_score_gemma":0.00635845,"domain_scores_codex":[0.9972128,0.001617581,0.0002470075,0.0005325134,0.0002732049,0.000116949],"domain_scores_gemma":[0.9832591,0.01379226,0.0005079204,0.0007023882,0.001445,0.000293426],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003712418,0.0008791186,0.01062725,0.001380241,0.001566835,0.0004612286,0.0007401545,0.1522861,0.02495134,0.001775609,0.05325057,0.7483692],"study_design_scores_gemma":[0.0002382022,0.0004884746,0.003555301,0.00008113742,0.0006406442,0.0001714362,0.0001745096,0.9717281,0.008998472,0.007135444,0.006720969,0.0000672688],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2112013,0.01315701,0.7369706,0.006000019,0.001673517,0.0006040697,0.007028454,0.01721436,0.006150641],"genre_scores_gemma":[0.7999325,0.002700444,0.1680123,0.001730043,0.001757392,0.0003673864,0.01747366,0.000920566,0.007105789],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005629766,"threshold_uncertainty_score":0.02977341,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02209648554058223,"score_gpt":0.3445173154923743,"score_spread":0.3224208299517921,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}