{"id":"W3030026364","doi":"10.1145/3318464.3380568","title":"SCODED: Statistical Constraint Oriented Data Error Detection","year":2020,"lang":"en","type":"article","venue":"","topic":"Data Stream Mining Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Leverage (statistics); Computer science; Constraint (computer-aided design); Data mining; Statistical model; Key (lock); Data integrity; Error detection and correction; Data modeling; Algorithm; Machine learning; Database; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01233973,0.001842609,0.001810355,0.005821861,0.001559842,0.003756311,0.004772193,0.001816242,0.002620461],"category_scores_gemma":[0.06975247,0.0009621915,0.002044062,0.0059373,0.00216784,0.005226282,0.006160152,0.003789401,0.001423536],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001457151,"about_ca_system_score_gemma":0.006619785,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006643284,"about_ca_topic_score_gemma":0.008696856,"domain_scores_codex":[0.9848878,0.002886673,0.002107606,0.002428614,0.007133772,0.0005555116],"domain_scores_gemma":[0.9288532,0.0339185,0.008069137,0.01503035,0.01317354,0.0009551996],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001104819,0.0004252129,0.04822168,0.001905905,0.0009284855,0.001365109,0.001434673,0.0765915,0.02397501,0.05282503,0.04145477,0.7497678],"study_design_scores_gemma":[0.0001147932,0.000271501,0.005412639,0.0003211179,0.000179876,0.001053862,0.0004859538,0.8272612,0.06244958,0.06598275,0.03626643,0.0002003023],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01033602,0.0003865694,0.9701792,0.0006629218,0.0001158385,0.0004058205,0.002123291,0.01485545,0.000934898],"genre_scores_gemma":[0.1511681,0.0004238116,0.8387986,0.0008038121,0.0001732978,0.0005183054,0.005501305,0.001267848,0.001344956],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01233973,"threshold_uncertainty_score":0.06525952,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08881471810412497,"score_gpt":0.3212253166233988,"score_spread":0.2324105985192738,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}