{"id":"W4293058541","doi":"10.21203/rs.3.rs-1766545/v1","title":"Modeling electronic health record data using an end-to-end knowledge-graph-informed topic model","year":2022,"lang":"en","type":"preprint","venue":"Research Square","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Knowledge graph; Embedding; Graph; Health records; Information retrieval; Data mining; End-to-end principle; Imputation (statistics); Machine learning; Data science; Artificial intelligence; Missing data; Theoretical computer science; Health care","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001802137,0.0008859116,0.0008467347,0.001443067,0.0003911134,0.001174408,0.001897527,0.001922706,0.002495263],"category_scores_gemma":[0.006487166,0.0006029615,0.001683138,0.001833637,0.0005413108,0.001885071,0.001306838,0.002356773,0.00123857],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001266565,"about_ca_system_score_gemma":0.001215442,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01558266,"about_ca_topic_score_gemma":0.02450016,"domain_scores_codex":[0.9992687,0.0002746735,0.00004165579,0.000268724,0.00007683843,0.00006936031],"domain_scores_gemma":[0.9971514,0.002114316,0.0001673942,0.0002122034,0.000255652,0.0000990443],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000247075,0.0002027428,0.005317155,0.0001447384,0.0001495101,0.0001633474,0.0002275267,0.8621189,0.001223817,0.01219775,0.007467017,0.1105404],"study_design_scores_gemma":[0.000009433873,0.00001109412,0.0002137949,0.000006609456,0.00001092751,0.00001471374,0.00000885639,0.9920339,0.0001818679,0.006993629,0.0005106982,0.000004506136],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0369941,0.0005556523,0.9535401,0.001305728,0.00006078654,0.0001365805,0.003998084,0.002362101,0.001046894],"genre_scores_gemma":[0.6027626,0.0009177668,0.3696493,0.0007736579,0.0002365485,0.0006083645,0.01773616,0.0004059494,0.006909675],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01558266,"threshold_uncertainty_score":0.03098392,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3070581037583382,"score_gpt":0.5185502015219003,"score_spread":0.2114920977635621,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}