{"id":"W4407093098","doi":"10.48550/arxiv.2501.18891","title":"CAAT-EHR: Cross-Attentional Autoregressive Transformer for Multimodal Electronic Health Record Embeddings","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Canadian Institutes of Health Research; National Institutes of Health; Genentech; IXICO; H. Lundbeck A/S; Servier; Eisai; University of Southern California; Biogen; Eli Lilly and Company; Bristol-Myers Squibb; BioClinica; U.S. Department of Defense; Meso Scale Diagnostics; Alzheimer's Disease Neuroimaging Initiative; Pfizer; National Institute of General Medical Sciences; Northern California Institute for Research and Education; University of North Texas; Alzheimer's Association","keywords":"Autoregressive model; Transformer; Electronic health record; Health records; Computer science; Psychology; Econometrics; Engineering; Economics; Electrical engineering; Health care","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.001119021,0.000646178,0.0008422712,0.0003534333,0.0007125502,0.0002831829,0.002347623,0.0005611581,0.00007596752],"category_scores_gemma":[0.0002757878,0.0006618541,0.0005947944,0.0003012526,0.0001202911,0.0003334381,0.0006495732,0.002410705,0.00005782929],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001076079,"about_ca_system_score_gemma":0.003866147,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002242462,"about_ca_topic_score_gemma":0.0005327375,"domain_scores_codex":[0.9946958,0.0002871069,0.001061218,0.001885001,0.0005628247,0.001508005],"domain_scores_gemma":[0.9966542,0.0004276922,0.0007561306,0.001305024,0.0005441902,0.0003128394],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000176811,0.0002983601,0.8174896,0.004460074,0.000409342,0.00001806357,0.003839492,0.006636932,0.00005353673,0.02709538,0.004553169,0.1349693],"study_design_scores_gemma":[0.002846065,0.0009357699,0.5719857,0.001896463,0.00006821853,0.0000378589,0.00005100991,0.2884603,0.0002816836,0.01353437,0.1179414,0.001961212],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3855271,0.001826654,0.5722953,0.02902826,0.005636463,0.003390449,0.0003481979,0.001084225,0.0008633194],"genre_scores_gemma":[0.9290841,0.000318362,0.05172748,0.004117523,0.0009366007,0.001329553,0.0005106076,0.00009587483,0.01187995],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5435569,"threshold_uncertainty_score":0.9998907,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0359103846591871,"score_gpt":0.378476466764079,"score_spread":0.3425660821048919,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}