{"id":"W4392736665","doi":"10.1093/jamia/ocae312","title":"A dataset and benchmark for hospital course summarization with adapted large language models","year":2024,"lang":"en","type":"article","venue":"Journal of the American Medical Informatics Association","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Stanford Cardiovascular Institute, School of Medicine, Stanford University; National Institute of Biomedical Imaging and Bioengineering; National Institute of Arthritis and Musculoskeletal and Skin Diseases; Stanford Institute for Human-Centered Artificial Intelligence, Stanford University; National Institute of Nursing Research; National Heart, Lung, and Blood Institute; National Institutes of Health","keywords":"Computer science; Context (archaeology); Automatic summarization; Benchmark (surveying); Artificial intelligence; Benchmarking; Health care; Machine learning; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002451871,0.002630308,0.0007442542,0.002959457,0.0008820754,0.001224366,0.002981551,0.002506033,0.006094604],"category_scores_gemma":[0.01175776,0.0003448247,0.001758287,0.002482179,0.0005321296,0.001476727,0.001591033,0.001817401,0.006953512],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0022328,"about_ca_system_score_gemma":0.002295269,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01736439,"about_ca_topic_score_gemma":0.03034143,"domain_scores_codex":[0.9970482,0.0008382557,0.0005506561,0.0008004621,0.0005865288,0.0001758503],"domain_scores_gemma":[0.9958588,0.001639155,0.0003033783,0.0008498657,0.001099523,0.0002491543],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001614843,0.001232863,0.01026922,0.003818035,0.0004695191,0.001020148,0.0003968836,0.03021274,0.01331495,0.002258958,0.7378396,0.1975523],"study_design_scores_gemma":[0.002584721,0.002036812,0.03862703,0.000887715,0.0004956017,0.002197878,0.001451071,0.2344256,0.03958642,0.009579435,0.6676481,0.000479656],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.07599941,0.004920384,0.03314286,0.003016169,0.001266952,0.002092439,0.8296386,0.04047291,0.009450236],"genre_scores_gemma":[0.03442999,0.0004046074,0.03589396,0.0005071861,0.0001265571,0.0009023261,0.925126,0.000425396,0.00218389],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.01736439,"threshold_uncertainty_score":0.03452665,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.005396857601458518,"score_gpt":0.2874210560819696,"score_spread":0.2820241984805111,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}