{"id":"W3101513017","doi":"10.18653/v1/2020.clinicalnlp-1.15","title":"MeDAL: Medical Abbreviation Disambiguation Dataset for Natural Language Understanding Pretraining","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; McGill University","funders":"","keywords":"Computer science; Natural language processing; Medal; Artificial intelligence; Domain (mathematical analysis); Natural language; Natural (archaeology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002256438,0.001879368,0.0009209864,0.004124634,0.001056073,0.001017372,0.002201402,0.002973627,0.01235191],"category_scores_gemma":[0.01118445,0.0004191412,0.001285624,0.002168198,0.0006687702,0.001316325,0.001962371,0.002257302,0.01075193],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001278889,"about_ca_system_score_gemma":0.002711131,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006441092,"about_ca_topic_score_gemma":0.0188242,"domain_scores_codex":[0.9979903,0.0005775734,0.000373057,0.0006008208,0.0003405366,0.0001178297],"domain_scores_gemma":[0.9953323,0.002439652,0.0003571792,0.0008076669,0.0007083123,0.0003548859],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0007007403,0.0004089542,0.008025385,0.003306431,0.0002316577,0.001209622,0.0004687974,0.005082796,0.01061692,0.002827249,0.871819,0.09530247],"study_design_scores_gemma":[0.001240886,0.0005270647,0.0315196,0.0006534249,0.0002946739,0.005010434,0.0008216132,0.04476808,0.02875634,0.01084606,0.8753183,0.0002435176],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.04339947,0.004328857,0.03333676,0.003745955,0.0008151477,0.001296579,0.8861929,0.01802533,0.008859114],"genre_scores_gemma":[0.03383287,0.0005241134,0.04553868,0.000941171,0.0001727005,0.0009453306,0.9149882,0.0004560302,0.00260104],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.01235191,"threshold_uncertainty_score":0.04132122,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1476165516358032,"score_gpt":0.3630807248595844,"score_spread":0.2154641732237812,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}