{"id":"W4290944545","doi":"10.1145/3534678.3542675","title":"Automatic Phenotyping by a Seed-guided Topic Model","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Canada First Research Excellence Fund","keywords":"Inference; Computer science; Topic model; Machine learning; Artificial intelligence; Bayesian inference; Source code; Prior probability; Vocabulary; Natural language processing; Bayesian probability; Data science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["open_science"],"consensus_categories":["open_science"],"category_scores_codex":[0.0009157613,0.0002481511,0.0003281242,0.0001156337,0.0006024899,0.000387129,0.005638235,0.00005193763,0.00002160033],"category_scores_gemma":[0.0008876554,0.0002043861,0.00004991619,0.0004598094,0.00009724033,0.001462232,0.008747732,0.0005071893,0.000002715107],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006807524,"about_ca_system_score_gemma":0.0002841888,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003177988,"about_ca_topic_score_gemma":0.000003365019,"domain_scores_codex":[0.9978684,0.00006049945,0.0004350263,0.0008164048,0.0004296839,0.0003900165],"domain_scores_gemma":[0.9979837,0.0002388026,0.0003810319,0.00115592,0.000139789,0.0001007443],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00010323,0.0006821263,0.0597103,0.002463595,0.0001668795,0.000003464075,0.04319841,0.0006056217,0.01101933,0.6398805,0.02771922,0.2144473],"study_design_scores_gemma":[0.0003477298,0.0001109161,0.0009716559,0.0002676066,0.00002102109,0.00001755284,0.001086289,0.9880689,0.0004498821,0.007786026,0.0005885644,0.000283906],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.98037,0.0005904632,0.005428968,0.003990438,0.0004696442,0.0005035229,0.0002022199,0.0001928634,0.008251806],"genre_scores_gemma":[0.9896294,0.00001990408,0.008876101,0.0003499152,0.00004172314,0.00005319612,0.00002599388,0.00001765785,0.0009861228],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9874632,"threshold_uncertainty_score":0.9997417,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08150570998099894,"score_gpt":0.3294082098277484,"score_spread":0.2479024998467494,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}