{"id":"W4385550818","doi":"10.31219/osf.io/kn5f2","title":"Performance Analysis of Large Language Models for Medical Text Summarization","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Automatic summarization; Pace; Relevance (law); Computer science; Unified Medical Language System; Medical literature; Data science; Natural language processing; Medicine; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008625751,0.000132577,0.0003783951,0.000445802,0.00003925494,0.0000465448,0.001210634,0.0002408849,0.00004881539],"category_scores_gemma":[0.00009945496,0.0001190901,0.0002141946,0.0005923845,0.00001285055,0.0001530951,0.001393373,0.0001791199,0.000005507462],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003149304,"about_ca_system_score_gemma":0.0001721878,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001659774,"about_ca_topic_score_gemma":0.0002267755,"domain_scores_codex":[0.9982665,0.00002999841,0.0004279029,0.0005035335,0.0005507419,0.000221305],"domain_scores_gemma":[0.9986209,0.0001273898,0.0001583838,0.0008914149,0.0001313807,0.00007053575],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000004372,0.00006413296,0.001897898,0.0003905133,0.000677887,0.000002995575,0.002486385,0.8732997,0.00001946842,0.09377351,0.0004090613,0.02697409],"study_design_scores_gemma":[0.0001135259,0.000007726829,0.0004118961,0.00004383354,0.0001190403,1.386256e-7,0.00002746774,0.9973397,0.00009719565,0.001689308,0.00002896547,0.0001212339],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02643422,0.00004099695,0.9713465,0.0004301029,0.0003102777,0.0002124978,0.00003293315,0.0002092562,0.0009832142],"genre_scores_gemma":[0.9146111,0.00006650545,0.08319855,0.0001479867,0.00007072265,0.00005657285,0.0001866676,0.00001313661,0.001648752],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8881769,"threshold_uncertainty_score":0.4856351,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03542364612472997,"score_gpt":0.3056338770938207,"score_spread":0.2702102309690907,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}