{"id":"W4293058541","doi":"10.21203/rs.3.rs-1766545/v1","title":"Modeling electronic health record data using an end-to-end knowledge-graph-informed topic model","year":2022,"lang":"en","type":"preprint","venue":"Research Square","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Knowledge graph; Embedding; Graph; Health records; Information retrieval; Data mining; End-to-end principle; Imputation (statistics); Machine learning; Data science; Artificial intelligence; Missing data; Theoretical computer science; Health care","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","open_science","research_integrity"],"consensus_categories":["open_science"],"category_scores_codex":[0.01029186,0.0005703127,0.0008322823,0.001996685,0.001765534,0.0008505238,0.01146763,0.000351495,0.000194551],"category_scores_gemma":[0.0009783624,0.0006288605,0.0001785854,0.002154015,0.00007692171,0.0009085525,0.02617885,0.008236948,0.00004074318],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003798964,"about_ca_system_score_gemma":0.02143273,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01697736,"about_ca_topic_score_gemma":0.008297171,"domain_scores_codex":[0.9870413,0.003096798,0.001101129,0.002992542,0.002719918,0.00304829],"domain_scores_gemma":[0.9890332,0.0005044842,0.0002511402,0.008260082,0.0008236415,0.001127438],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003011939,0.0001613377,0.0005154347,0.001925552,0.00003723753,0.00001423323,0.00465515,0.8697592,0.000004977969,0.008429015,0.000543747,0.113924],"study_design_scores_gemma":[0.0002012721,0.0005370466,0.00008282115,0.0004905279,0.000004633293,0.00001079379,0.0003358515,0.9820038,0.000001734511,0.01034821,0.005463303,0.0005200022],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06489424,0.004378362,0.916826,0.006703942,0.001086779,0.003604341,0.0003384702,0.0007767048,0.001391155],"genre_scores_gemma":[0.8519108,0.001466731,0.1426525,0.0004840624,0.000738218,0.0004743973,0.001468448,0.0001748041,0.0006300246],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7870166,"threshold_uncertainty_score":0.9996163,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3070581037583382,"score_gpt":0.5185502015219003,"score_spread":0.2114920977635621,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}