{"id":"W1626378195","doi":"10.1145/1938551.1938585","title":"Data cleaning and query answering with matching dependencies and matching functions","year":2011,"lang":"en","type":"article","venue":"","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":69,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia; Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada; International Business Machines Corporation","keywords":"Tuple; Matching (statistics); Computer science; Monotone polygon; Conjunctive query; Dependency (UML); Functional dependency; Semantics (computer science); Dependency theory (database theory); Query optimization; Upper and lower bounds; Theoretical computer science; Mathematics; Information retrieval; Discrete mathematics; Relational database; Artificial intelligence; Programming language","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01475668,0.001067199,0.00211269,0.002814244,0.0015259,0.005284358,0.002940772,0.002600736,0.001985002],"category_scores_gemma":[0.04748325,0.001161838,0.003255594,0.004320249,0.004018411,0.01308179,0.005157213,0.004582676,0.0005119955],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002373951,"about_ca_system_score_gemma":0.002614689,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00266335,"about_ca_topic_score_gemma":0.001855685,"domain_scores_codex":[0.9808854,0.008164917,0.001608946,0.002631764,0.005406156,0.00130269],"domain_scores_gemma":[0.9612765,0.02759493,0.002495625,0.005575846,0.002454697,0.0006024205],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006304579,0.0004224866,0.004745035,0.0006689017,0.0002044959,0.0003470792,0.001188239,0.1841217,0.009360426,0.6639196,0.004725653,0.1296659],"study_design_scores_gemma":[0.00006460142,0.0001109613,0.0005742453,0.00003884728,0.00006940087,0.000365464,0.0002164055,0.5610893,0.008078706,0.4243236,0.005025012,0.00004336224],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01924519,0.0002836008,0.9775968,0.000819988,0.00001643509,0.0001121883,0.0001778546,0.0004201394,0.001327738],"genre_scores_gemma":[0.26604,0.0004104662,0.7296931,0.0005222438,0.000133462,0.0003346649,0.0008203544,0.0001835046,0.001862261],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01475668,"threshold_uncertainty_score":0.07804173,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3569999851508519,"score_gpt":0.3743921688869392,"score_spread":0.01739218373608725,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}