{"id":"W2356882517","doi":"10.1093/jamia/ocw028","title":"Learning statistical models of phenotypes using noisy labeled training data","year":2016,"lang":"en","type":"article","venue":"Journal of the American Medical Informatics Association","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":165,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"U.S. National Library of Medicine; National Institute of General Medical Sciences; National Human Genome Research Institute","keywords":"Computer science; Scalability; Machine learning; Artificial intelligence; Logistic regression; Feature (linguistics); Phenotype; Feature engineering; Implementation; Predictive modelling; Data mining; Deep learning; Database","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009586131,0.001305318,0.0010124,0.001873402,0.0004992682,0.001891887,0.001722261,0.001156445,0.0008825135],"category_scores_gemma":[0.04012385,0.0005809961,0.001133878,0.001229683,0.001106407,0.001743907,0.001199148,0.001733784,0.0005646524],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001432088,"about_ca_system_score_gemma":0.001719628,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003888665,"about_ca_topic_score_gemma":0.006047883,"domain_scores_codex":[0.994936,0.002707852,0.0003258562,0.001321068,0.0005832041,0.0001261182],"domain_scores_gemma":[0.9611462,0.02955842,0.003204297,0.003392693,0.002368602,0.0003298941],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002744648,0.000330838,0.04375922,0.0001864257,0.0002939466,0.000339393,0.0004146907,0.8568134,0.00196855,0.006200058,0.003307969,0.08611118],"study_design_scores_gemma":[0.00002469483,0.00004997397,0.002265082,0.00003603415,0.00002729475,0.00004500444,0.00004292846,0.9855378,0.0008962569,0.01054968,0.0005133454,0.0000119185],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1018757,0.0001848363,0.8935858,0.000583356,0.00002943502,0.0001464031,0.001460343,0.001462473,0.0006716987],"genre_scores_gemma":[0.6732357,0.0001691689,0.3173729,0.0004080855,0.00006600975,0.0006786002,0.007042218,0.0001592437,0.0008681259],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009586131,"threshold_uncertainty_score":0.05069691,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04244371233329072,"score_gpt":0.3199957147633697,"score_spread":0.277552002430079,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}