{"id":"W3021671533","doi":"10.2196/15852","title":"The Development of the Military Service Identification Tool: Identifying Military Veterans in a Clinical Research Database Using Natural Language Processing and Machine Learning","year":2020,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"King's College London; Engineering and Physical Sciences Research Council; National Institute for Health and Care Research; NIHR Biomedical Research Centre, Royal Marsden NHS Foundation Trust/Institute of Cancer Research; Department of Health and Social Care; Institute of Psychiatry, Psychology and Neuroscience, King’s College London; Forces in Mind Trust","keywords":"SQL; Computer science; Identification (biology); Artificial intelligence; Service (business); Database; Machine learning; Medicine; Natural language processing; Information retrieval; Data mining","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006554854,0.0001212147,0.0001977673,0.00008407558,0.0005989311,0.0000753368,0.00130263,0.0001019242,0.000003235],"category_scores_gemma":[0.002540071,0.00007691044,0.00003802451,0.0009730308,0.0001889639,0.0006400062,0.00132627,0.001996911,0.000004694215],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005941903,"about_ca_system_score_gemma":0.0005618578,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002655124,"about_ca_topic_score_gemma":0.0006614055,"domain_scores_codex":[0.9958493,0.0007932094,0.001455042,0.0002081454,0.001348869,0.0003454244],"domain_scores_gemma":[0.9982284,0.0008036447,0.0002004118,0.000413901,0.0001916051,0.000162003],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005846167,0.00006297661,0.03066946,0.003826731,0.00002155579,0.00001486003,0.3610429,0.0003626243,0.0002710069,0.0001296095,0.00006456556,0.6034752],"study_design_scores_gemma":[0.0002872862,0.00001904901,0.02749045,0.0004804278,0.000002386867,0.00001212989,0.006931222,0.9641938,0.00004432501,0.00001314434,0.0004397062,0.00008614389],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9776168,0.002595511,0.01510906,0.004018784,0.0001668542,0.0004272499,0.000002180084,0.00004972989,0.00001384739],"genre_scores_gemma":[0.9722782,0.00006100714,0.02661956,0.0009345248,0.00005946814,0.0000176353,0.00001434366,0.000008780622,0.000006455305],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9638311,"threshold_uncertainty_score":0.8675694,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1320546192037357,"score_gpt":0.4406966219772892,"score_spread":0.3086420027735535,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}