{"id":"W2544486974","doi":"10.14778/2994509.2994518","title":"Detecting data errors","year":2016,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":237,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"University of California Berkeley","keywords":"Computer science; Raw data; Outlier; Data mining; Set (abstract data type); Variety (cybernetics); Data quality; Ground truth; Anomaly detection; Quality (philosophy); Big data; Data science; Machine learning; Artificial intelligence; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02098097,0.002211681,0.001998074,0.009275529,0.001645167,0.005104362,0.004321513,0.002830164,0.002670607],"category_scores_gemma":[0.1517903,0.000865113,0.002016069,0.007719268,0.001597924,0.005799147,0.006564772,0.002727047,0.002860892],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001315207,"about_ca_system_score_gemma":0.003249373,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002358616,"about_ca_topic_score_gemma":0.002255817,"domain_scores_codex":[0.9522556,0.009320097,0.007339614,0.01126091,0.01824003,0.00158373],"domain_scores_gemma":[0.7991547,0.07963275,0.01970252,0.06453487,0.03549375,0.001481378],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009512166,0.0005889611,0.1189586,0.003738887,0.0007083282,0.002322092,0.008237539,0.01397714,0.03586598,0.02107788,0.04859595,0.7449773],"study_design_scores_gemma":[0.0002018019,0.0008980662,0.07099519,0.002114834,0.0008785389,0.004792019,0.008833255,0.21652,0.3119785,0.07103218,0.3110456,0.000709948],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.117291,0.002201726,0.810873,0.003246109,0.0008943844,0.001943357,0.01510004,0.03976662,0.008683854],"genre_scores_gemma":[0.2672631,0.0006566591,0.7069734,0.001401538,0.0001439539,0.0009840429,0.01583991,0.002757099,0.003980435],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02098097,"threshold_uncertainty_score":0.1109593,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3097609978046282,"score_gpt":0.4126229492891771,"score_spread":0.1028619514845489,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}