{"id":"W2949244208","doi":"10.48550/arxiv.1702.00820","title":"HoloClean: Holistic Data Repairs with Probabilistic Inference","year":2017,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":57,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Inference; Leverage (statistics); Probabilistic logic; Tuple; Computer science; Data mining; Statistical model; Machine learning; Artificial intelligence; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006983992,0.001858686,0.001934413,0.003497685,0.001063872,0.003785878,0.006570341,0.001806547,0.00683478],"category_scores_gemma":[0.03693419,0.001402398,0.004149397,0.002624025,0.002675645,0.006180714,0.008092601,0.004132321,0.002944331],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002161958,"about_ca_system_score_gemma":0.005466596,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007520391,"about_ca_topic_score_gemma":0.01283115,"domain_scores_codex":[0.9927712,0.001520391,0.0005263504,0.002011495,0.002832095,0.0003385158],"domain_scores_gemma":[0.9860638,0.00551377,0.001076944,0.005601836,0.001450049,0.0002935401],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004651978,0.0002173624,0.008345112,0.001421845,0.0004674071,0.0005117674,0.0008443043,0.2643736,0.007827181,0.06220427,0.05054162,0.6027803],"study_design_scores_gemma":[0.00006196211,0.0000825638,0.0007000476,0.0001385627,0.00007856025,0.000358944,0.0001415548,0.8605657,0.01151706,0.09788354,0.02840046,0.00007099222],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.002085466,0.0002787326,0.9797173,0.0003319088,0.00005632043,0.0001212051,0.0009212209,0.01571471,0.0007731129],"genre_scores_gemma":[0.06430685,0.0003146104,0.9256713,0.0004444522,0.00007262766,0.0002820803,0.004671474,0.002346955,0.001889575],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007520391,"threshold_uncertainty_score":0.03693533,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.631797409494457,"score_gpt":0.3557862297306885,"score_spread":0.2760111797637684,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}