{"id":"W4367312293","doi":"10.2196/42721","title":"Estimating Rare Disease Incidences With Large-scale Internet Search Data: Development and Evaluation of a Two-step Machine Learning Method","year":2023,"lang":"en","type":"article","venue":"JMIR Infodemiology","topic":"Data-Driven Disease Surveillance","field":"Medicine","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Peking Union Medical College; Peking Union Medical College Hospital; Tsinghua University; National Natural Science Foundation of China","keywords":"Computer science; Search engine indexing; Matching (statistics); The Internet; Online search; Scale (ratio); Session (web analytics); Information retrieval; Web search query; Big data; Search engine; Medicine; Machine learning; Data mining; Statistics; Geography; World Wide Web; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004957099,0.0001520953,0.0004222026,0.0001664129,0.00006587789,0.00001060725,0.0002121519,0.00005449654,0.00006910357],"category_scores_gemma":[0.001508899,0.0001178741,0.00002215418,0.0002954013,0.000112957,0.0001562595,0.0007319484,0.0002749731,0.00002230834],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005701617,"about_ca_system_score_gemma":0.000401128,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000109169,"about_ca_topic_score_gemma":0.0002473348,"domain_scores_codex":[0.9975931,0.0007600383,0.0004274681,0.0004684429,0.0004487849,0.0003021624],"domain_scores_gemma":[0.9984059,0.000509708,0.0001876338,0.0004625503,0.0002409071,0.0001932791],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003925326,0.00004760331,0.9610087,0.0003166426,0.00008591274,0.00003266834,0.00150177,0.009287914,0.00005936896,0.00005133252,0.0003460713,0.02686949],"study_design_scores_gemma":[0.001169573,0.00006798775,0.3831306,0.0001671481,0.00006311741,0.00001508975,0.0002101454,0.6145225,0.00001806206,0.00004357497,0.0005114044,0.00008080127],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9786473,0.0003281269,0.0196713,0.0002672941,0.00004399067,0.0006358053,0.0001432501,0.0001328539,0.0001301053],"genre_scores_gemma":[0.8788062,0.000009315931,0.1179377,0.0001270552,0.00004370178,0.0001326753,0.002847429,0.0000169598,0.00007890317],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6052346,"threshold_uncertainty_score":0.4806765,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1035949761319255,"score_gpt":0.4411465323995019,"score_spread":0.3375515562675763,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}