{"id":"W2995438613","doi":"10.48550/arxiv.1912.06174","title":"Training without training data: Improving the generalizability of automated medical abbreviation disambiguation","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Generalizability theory; Computer science; Context (archaeology); Artificial intelligence; Training set; Natural language processing; Labeled data; Machine learning; Representation (politics); Training (meteorology); Scarcity; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01002991,0.00210493,0.001686807,0.002392345,0.0009378797,0.002097509,0.002958012,0.002567159,0.001574043],"category_scores_gemma":[0.03431628,0.0007841759,0.00162858,0.002391116,0.001209104,0.004556637,0.003336052,0.003492112,0.001532274],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001089302,"about_ca_system_score_gemma":0.002064189,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01310387,"about_ca_topic_score_gemma":0.01418068,"domain_scores_codex":[0.99448,0.002335798,0.0004060127,0.001955016,0.0006025837,0.0002205199],"domain_scores_gemma":[0.9793251,0.01324169,0.0008594016,0.00470697,0.001580027,0.0002868881],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001391676,0.0006313268,0.03910427,0.0005113425,0.0006095716,0.0006708653,0.0006528673,0.2723045,0.01089473,0.002646203,0.02839554,0.6421872],"study_design_scores_gemma":[0.00008570204,0.0001309205,0.004585824,0.00008007964,0.0001191429,0.0001719841,0.0001463777,0.9758301,0.005225713,0.006981984,0.006599824,0.00004226643],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3124644,0.005725542,0.6390265,0.005226135,0.0006442094,0.000591848,0.005442369,0.02444721,0.006431793],"genre_scores_gemma":[0.7852929,0.001060281,0.1946851,0.001818264,0.0003785515,0.0003223762,0.01326065,0.0006601408,0.002521758],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01310387,"threshold_uncertainty_score":0.0530439,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1841075146981282,"score_gpt":0.2660201738176082,"score_spread":0.08191265911947995,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}