{"id":"W4414270172","doi":"10.1101/2025.09.12.675926","title":"Medical Abbreviation Disambiguation with Large Language Models: Zero- and Few-Shot Evaluation on the MeDAL Dataset","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Brock University","funders":"","keywords":"Interpretability; Readability; Task (project management); Unified Medical Language System; Language model; Resource (disambiguation); Named entity; Information extraction","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01007728,0.002299072,0.001913413,0.002602693,0.001443331,0.002171242,0.003313482,0.003483702,0.002558058],"category_scores_gemma":[0.02411532,0.0005259896,0.00180475,0.001538307,0.001659946,0.004314995,0.003007518,0.002945801,0.00172022],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001855048,"about_ca_system_score_gemma":0.002841213,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01442323,"about_ca_topic_score_gemma":0.02211652,"domain_scores_codex":[0.9943413,0.00267121,0.0005120526,0.001474765,0.0007518572,0.0002487606],"domain_scores_gemma":[0.9829214,0.01312674,0.0003685763,0.001660314,0.001188956,0.0007339753],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.006753265,0.004149744,0.01193107,0.004714977,0.002339435,0.0026138,0.002046379,0.2216596,0.0296195,0.00514475,0.1129501,0.5960774],"study_design_scores_gemma":[0.0006542421,0.00131871,0.005165262,0.000219403,0.0004614654,0.00118113,0.001445282,0.9413854,0.02516992,0.008166536,0.01462151,0.0002111679],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7870056,0.0172855,0.1149415,0.004458428,0.002065729,0.0008979706,0.01761016,0.04725093,0.008484268],"genre_scores_gemma":[0.75705,0.002020075,0.1693261,0.002445217,0.0004669784,0.0004457911,0.06140898,0.001476382,0.005360517],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01442323,"threshold_uncertainty_score":0.05329442,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02525283076595809,"score_gpt":0.2839559906917193,"score_spread":0.2587031599257612,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}