{"id":"W3021671533","doi":"10.2196/15852","title":"The Development of the Military Service Identification Tool: Identifying Military Veterans in a Clinical Research Database Using Natural Language Processing and Machine Learning","year":2020,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"King's College London; Engineering and Physical Sciences Research Council; National Institute for Health and Care Research; NIHR Biomedical Research Centre, Royal Marsden NHS Foundation Trust/Institute of Cancer Research; Department of Health and Social Care; Institute of Psychiatry, Psychology and Neuroscience, King’s College London; Forces in Mind Trust","keywords":"SQL; Computer science; Identification (biology); Artificial intelligence; Service (business); Database; Machine learning; Medicine; Natural language processing; Information retrieval; Data mining","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00918531,0.0007769957,0.000678377,0.004468261,0.0007312273,0.002630045,0.001509385,0.0009607482,0.001661059],"category_scores_gemma":[0.02571601,0.0004019685,0.0008999727,0.001835488,0.0005772879,0.002639782,0.00176192,0.001581645,0.001656416],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001274158,"about_ca_system_score_gemma":0.004347265,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006716567,"about_ca_topic_score_gemma":0.008987633,"domain_scores_codex":[0.9943382,0.002274241,0.0008603075,0.001089817,0.001230382,0.0002070075],"domain_scores_gemma":[0.9831744,0.01108814,0.001072981,0.0009418506,0.003174132,0.0005485248],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005844394,0.0009020575,0.03525511,0.001936872,0.0002260954,0.00143986,0.003491541,0.01353113,0.03982074,0.005597471,0.03666576,0.8605489],"study_design_scores_gemma":[0.0002913749,0.001575421,0.03654017,0.0008849782,0.0002493122,0.002837891,0.004935989,0.792335,0.07830554,0.01071702,0.07105303,0.0002743178],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1703171,0.000802761,0.7599589,0.003880938,0.000183925,0.004665834,0.01102378,0.04607841,0.0030884],"genre_scores_gemma":[0.1230074,0.0001724081,0.866263,0.0004199948,0.00004125855,0.0009052663,0.008158632,0.0002694213,0.000762444],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00918531,"threshold_uncertainty_score":0.04857719,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1320546192037357,"score_gpt":0.4406966219772892,"score_spread":0.3086420027735535,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}