{"id":"W2965327127","doi":"10.1093/jamia/ocz112","title":"Development of a global infectious disease activity database using natural language processing, machine learning, and human expertise","year":2019,"lang":"en","type":"article","venue":"Journal of the American Medical Informatics Association","topic":"Data-Driven Disease Surveillance","field":"Medicine","cited_by":31,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; BlueDot (Canada); St. Michael's Hospital","funders":"","keywords":"Computer science; Machine learning; Artificial intelligence; Infectious disease (medical specialty); Representativeness heuristic; Classifier (UML); Information extraction; Natural language processing; Disease; Medicine; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008561022,0.000902816,0.001023918,0.007976286,0.000605256,0.002362289,0.001415005,0.0007937182,0.00132007],"category_scores_gemma":[0.02065057,0.0003798934,0.0008020457,0.002956203,0.000411904,0.003297307,0.001546965,0.000905893,0.001133045],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001511524,"about_ca_system_score_gemma":0.003085048,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0118365,"about_ca_topic_score_gemma":0.01172411,"domain_scores_codex":[0.9951278,0.001353854,0.0009115976,0.001318783,0.001116529,0.0001715394],"domain_scores_gemma":[0.9822888,0.007297669,0.001737428,0.002414674,0.005507162,0.000754375],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006837667,0.001969254,0.1792925,0.001251178,0.0004759217,0.001071981,0.001306787,0.02119452,0.03769532,0.002070522,0.03441988,0.7185684],"study_design_scores_gemma":[0.0003471348,0.001770359,0.152254,0.0004151986,0.0004546453,0.001260224,0.002173089,0.6739021,0.101054,0.005365513,0.06073619,0.0002675457],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4653707,0.001316648,0.4324315,0.00287281,0.0002935517,0.005882087,0.05075738,0.03368025,0.007395119],"genre_scores_gemma":[0.4163519,0.0003252852,0.5168261,0.000405113,0.0001289762,0.001604938,0.06274127,0.0002832806,0.001333075],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0118365,"threshold_uncertainty_score":0.04527551,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00696299589376862,"score_gpt":0.3142278333592521,"score_spread":0.3072648374654834,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}