{"id":"W2295468252","doi":"10.14778/2856318.2856325","title":"Combining quantitative and logical data cleaning","year":2015,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":104,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ontario Tech University; McMaster University; University of Toronto","funders":"","keywords":"Computer science; Metric (unit); Inference; Functional dependency; Set (abstract data type); Distortion (music); Statistical inference; Data mining; Algorithm; Theoretical computer science; Dependency (UML); Quality (philosophy); Data quality; Artificial intelligence; Relational database; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01963293,0.001763221,0.002002778,0.005336357,0.001624887,0.005092177,0.006389302,0.002015429,0.002566699],"category_scores_gemma":[0.05328358,0.001416168,0.003621856,0.005225926,0.004657819,0.009263154,0.01152399,0.004628743,0.0009114493],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002409616,"about_ca_system_score_gemma":0.004903195,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003013635,"about_ca_topic_score_gemma":0.004353407,"domain_scores_codex":[0.9740335,0.007640894,0.002531967,0.003468806,0.01144813,0.0008767135],"domain_scores_gemma":[0.9354174,0.03012721,0.003291831,0.0236405,0.006916509,0.0006065988],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004300622,0.0004835255,0.01062826,0.001454712,0.0005501051,0.0007005003,0.001254775,0.1696998,0.02619185,0.1658238,0.01186396,0.6109186],"study_design_scores_gemma":[0.00008481284,0.000249364,0.002097468,0.0002416865,0.0002177236,0.0009744816,0.0008237375,0.6208883,0.04097774,0.3036481,0.02963696,0.0001596626],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003592829,0.0001494208,0.9926888,0.0006299151,0.0000323675,0.0001054785,0.0001743955,0.00170846,0.0009182467],"genre_scores_gemma":[0.08399864,0.0001624015,0.9131966,0.0004791341,0.00004791344,0.0001651064,0.0007536365,0.0004406451,0.0007557734],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01963293,"threshold_uncertainty_score":0.1038301,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6149975953066957,"score_gpt":0.4683191180089634,"score_spread":0.1466784772977324,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}