{"id":"W4403289769","doi":"10.1186/s12982-024-00245-3","title":"Efficient detection of data entry errors in large-scale public health surveys: an unsupervised machine learning approach","year":2024,"lang":"en","type":"article","venue":"Discover Public Health","topic":"Artificial Intelligence in Healthcare","field":"Health Professions","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ontario Ministry of Labour","funders":"Ministry of Health and Family Welfare; Indian Council of Medical Research","keywords":"Scale (ratio); Computer science; Unsupervised learning; Machine learning; Public health; Artificial intelligence; Data science; Medicine; Geography; Cartography; Nursing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.007913313,0.0008297801,0.001280328,0.003193547,0.0008042012,0.001470653,0.002070729,0.001077699,0.0002860721],"category_scores_gemma":[0.02955508,0.0004229483,0.000865687,0.002372531,0.0007612258,0.001568025,0.001314107,0.001488966,0.0003360049],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009126654,"about_ca_system_score_gemma":0.002648597,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008278782,"about_ca_topic_score_gemma":0.009349479,"domain_scores_codex":[0.9937788,0.002776915,0.0007193806,0.001344553,0.001091715,0.000288678],"domain_scores_gemma":[0.9705404,0.01682404,0.004443567,0.003261822,0.004607255,0.0003228431],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005098152,0.0007929985,0.1256865,0.0004886917,0.0004967635,0.0002469633,0.0007600811,0.1659983,0.007328727,0.004805895,0.005738926,0.6871462],"study_design_scores_gemma":[0.00002057015,0.0001051977,0.01615821,0.00005550413,0.00003890279,0.0001369701,0.0002705338,0.9690107,0.005217732,0.007306363,0.001640316,0.00003898479],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1047035,0.0003751011,0.8896248,0.0005661594,0.00005984894,0.00037839,0.0006574684,0.002808331,0.0008264175],"genre_scores_gemma":[0.5905001,0.0001681421,0.4063285,0.0002124835,0.00006914476,0.0002917997,0.001785211,0.00009509436,0.0005496438],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9920867,"threshold_uncertainty_score":0.04185009,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2832986354967932,"score_gpt":0.467259722435846,"score_spread":0.1839610869390528,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}