{"id":"W4283687125","doi":"10.2196/37557","title":"Automatic International Classification of Diseases Coding System: Deep Contextualized Language Model With Rule-Based Approaches","year":2022,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Medical Coding and Health Information","field":"Health Professions","cited_by":23,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Far Eastern Memorial Hospital; Ministry of Science and Technology, Taiwan","keywords":"Computer science; Medical diagnosis; Artificial intelligence; Medical classification; Diagnosis code; Unified Medical Language System; Preprocessor; Word2vec; Coding (social sciences); Language model; Natural language processing; Word embedding; Data mining; Machine learning; Medicine; Embedding","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002067938,0.00125202,0.0008656148,0.002461203,0.0004165217,0.001056404,0.001783474,0.0009043589,0.002593662],"category_scores_gemma":[0.008939856,0.000328305,0.001443931,0.001645517,0.0003727848,0.001383509,0.001219634,0.002193961,0.002187327],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001516331,"about_ca_system_score_gemma":0.002262725,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01992146,"about_ca_topic_score_gemma":0.01990966,"domain_scores_codex":[0.9979429,0.0006290438,0.0002615108,0.0006737115,0.0002996809,0.0001932093],"domain_scores_gemma":[0.996743,0.001546441,0.0003138495,0.0003048934,0.0009882059,0.0001036814],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005442529,0.0005110286,0.01267531,0.0005437462,0.0002535041,0.0003787878,0.000310224,0.1555355,0.004637681,0.008522015,0.04182516,0.7742627],"study_design_scores_gemma":[0.00002513014,0.00006492434,0.001873328,0.00006116389,0.0000408039,0.00007973365,0.00005195052,0.9815756,0.001863097,0.01141907,0.002914993,0.00003023435],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07472982,0.002130212,0.8816686,0.002283229,0.0006291615,0.000969845,0.01974704,0.01399587,0.003846186],"genre_scores_gemma":[0.4753548,0.0009503319,0.4732617,0.001166352,0.000318191,0.001526191,0.04302607,0.0003749624,0.004021484],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01992146,"threshold_uncertainty_score":0.03961098,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1839677208632765,"score_gpt":0.4185987175406165,"score_spread":0.23463099667734,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}