{"id":"W2591700809","doi":"10.14778/3137628.3137631","title":"HoloClean","year":2017,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":454,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency","keywords":"Leverage (statistics); Computer science; Probabilistic logic; Inference; Tuple; Data mining; Statistical model; Machine learning; Artificial intelligence; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004578141,0.001504516,0.001491478,0.00361551,0.001016351,0.005033804,0.006821305,0.001534417,0.01789622],"category_scores_gemma":[0.02542172,0.001209874,0.003122862,0.00255591,0.001827585,0.006923518,0.007741638,0.003278982,0.00826311],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00213188,"about_ca_system_score_gemma":0.005024464,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008731898,"about_ca_topic_score_gemma":0.01582726,"domain_scores_codex":[0.9949383,0.0007989898,0.0003875704,0.001465794,0.002102678,0.0003067198],"domain_scores_gemma":[0.9922168,0.002418151,0.0004580509,0.003561347,0.001102045,0.0002435548],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005564808,0.0002067958,0.006176967,0.001574929,0.0003254147,0.0004643956,0.000695684,0.06786154,0.004960692,0.1164097,0.1236348,0.6771326],"study_design_scores_gemma":[0.0001088519,0.0001124345,0.001084319,0.0003362753,0.00009390851,0.0006494065,0.0001737318,0.4664161,0.01203458,0.2181794,0.3006937,0.0001172597],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002447318,0.0008144128,0.9284461,0.0006306557,0.0001884736,0.0002584502,0.003887755,0.05710077,0.006226082],"genre_scores_gemma":[0.0565011,0.0006951344,0.9088446,0.001032132,0.0001207292,0.0004492227,0.01686006,0.007929657,0.007567351],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01789622,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2284211371904144,"score_gpt":0.4286142488382545,"score_spread":0.2001931116478401,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}