{"id":"W4367312293","doi":"10.2196/42721","title":"Estimating Rare Disease Incidences With Large-scale Internet Search Data: Development and Evaluation of a Two-step Machine Learning Method","year":2023,"lang":"en","type":"article","venue":"JMIR Infodemiology","topic":"Data-Driven Disease Surveillance","field":"Medicine","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Peking Union Medical College; Peking Union Medical College Hospital; Tsinghua University; National Natural Science Foundation of China","keywords":"Computer science; Search engine indexing; Matching (statistics); The Internet; Online search; Scale (ratio); Session (web analytics); Information retrieval; Web search query; Big data; Search engine; Medicine; Machine learning; Data mining; Statistics; Geography; World Wide Web; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01226428,0.001640572,0.002015126,0.005666248,0.0007849471,0.001742377,0.003047863,0.002071273,0.001945201],"category_scores_gemma":[0.02912398,0.0006823436,0.002233945,0.003634094,0.0006789723,0.002318301,0.001818895,0.002685991,0.0008833854],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001050983,"about_ca_system_score_gemma":0.002450602,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01403026,"about_ca_topic_score_gemma":0.009251268,"domain_scores_codex":[0.9947245,0.002442254,0.0006873239,0.001186584,0.0007205567,0.0002387615],"domain_scores_gemma":[0.9739691,0.02030293,0.001437888,0.001066667,0.002733242,0.0004902173],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008852899,0.001508806,0.2548714,0.0008559822,0.001431368,0.0005321639,0.0003475972,0.2895727,0.001762119,0.003246439,0.007564433,0.4374217],"study_design_scores_gemma":[0.00003019616,0.00007978736,0.005352355,0.0000285291,0.00006519081,0.00006071424,0.00005149121,0.992665,0.0002552287,0.001052253,0.0003386788,0.00002052567],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1767843,0.002187695,0.8108879,0.001421251,0.0002821843,0.001428486,0.002971872,0.002385574,0.001650859],"genre_scores_gemma":[0.5638955,0.0009627932,0.4245816,0.0005434218,0.0003643135,0.001685072,0.006050661,0.0001010874,0.001815461],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01403026,"threshold_uncertainty_score":0.06486052,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1035949761319255,"score_gpt":0.4411465323995019,"score_spread":0.3375515562675763,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}