{"id":"W4323539565","doi":"10.21203/rs.3.rs-2640617/v1","title":"Cerebrovascular disease case identification in inpatient electronic medical record data using natural language processing","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Alberta Health Services; University of Calgary","funders":"Canadian Institutes of Health Research","keywords":"Chart; Medical record; Medicine; Electronic medical record; Electronic health record; Coding (social sciences); Artificial intelligence; Predictive value; Clinical decision support system; Identification (biology); Medical diagnosis; Computer science; Natural language processing; Machine learning; Medical emergency; Statistics; Internal medicine; Decision support system; Pathology; Health care","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00305965,0.0002242283,0.0002553147,0.0003062079,0.0001688389,0.0001293573,0.001110141,0.0005408782,0.00002257982],"category_scores_gemma":[0.004394346,0.0002059167,0.00009612345,0.0004122471,0.0002928487,0.000007667323,0.003684326,0.00163694,0.00001487552],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001863121,"about_ca_system_score_gemma":0.00187765,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001295955,"about_ca_topic_score_gemma":0.002739662,"domain_scores_codex":[0.9958385,0.0006342614,0.0004383228,0.001185595,0.001124695,0.0007786805],"domain_scores_gemma":[0.9977747,0.00007595455,0.0001110302,0.001555907,0.0002105517,0.0002718696],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006385178,0.0009374283,0.03007915,0.007672276,0.0004089107,0.0127179,0.001516758,0.0003753368,0.009714908,0.00002702691,0.006750786,0.929161],"study_design_scores_gemma":[0.005322177,0.001065992,0.05070142,0.01247601,0.0003264128,0.001503565,0.01557667,0.8651331,0.005731025,0.001938591,0.03612723,0.004097791],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9802606,0.01678281,0.00124658,0.000595552,0.0003033588,0.0005741305,0.0001652065,0.00006213925,0.000009692283],"genre_scores_gemma":[0.9940147,0.001019634,0.0003745604,0.00002661618,0.0004384613,0.00009473973,0.003766444,0.00005092512,0.0002139301],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9250632,"threshold_uncertainty_score":0.8397039,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1224798412305595,"score_gpt":0.4628552906587924,"score_spread":0.3403754494282328,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}