{"id":"W4385550818","doi":"10.31219/osf.io/kn5f2","title":"Performance Analysis of Large Language Models for Medical Text Summarization","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Automatic summarization; Pace; Relevance (law); Computer science; Unified Medical Language System; Medical literature; Data science; Natural language processing; Medicine; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009532196,0.002494972,0.001628858,0.003280871,0.0007285341,0.001711642,0.00161935,0.001893866,0.001841204],"category_scores_gemma":[0.03078142,0.0004812392,0.001530392,0.002295297,0.0004723645,0.002751501,0.001071266,0.0019401,0.00173742],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001754721,"about_ca_system_score_gemma":0.002075619,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01606909,"about_ca_topic_score_gemma":0.0154933,"domain_scores_codex":[0.9952564,0.002632704,0.000580318,0.0008254871,0.0004407765,0.0002643473],"domain_scores_gemma":[0.9767104,0.01814943,0.0008245356,0.001383258,0.002370274,0.0005620713],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005225803,0.001203148,0.01277823,0.002483752,0.001981374,0.0006549786,0.0006244315,0.4792976,0.02208938,0.001773591,0.04455186,0.4273358],"study_design_scores_gemma":[0.000158286,0.0006222838,0.002387254,0.00005144158,0.0002608349,0.0001045873,0.0001614322,0.9832178,0.009133682,0.001418401,0.002426113,0.00005786127],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6642572,0.02424107,0.2258435,0.005522942,0.002097622,0.001037987,0.02043073,0.05015372,0.006415199],"genre_scores_gemma":[0.8455234,0.002136234,0.1119567,0.000645989,0.0005290581,0.0004553478,0.03508117,0.0007683379,0.00290389],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01606909,"threshold_uncertainty_score":0.0504117,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03542364612472997,"score_gpt":0.3056338770938207,"score_spread":0.2702102309690907,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}