{"id":"W2549035799","doi":"10.14778/3007263.3007320","title":"Qualitative data cleaning","year":2016,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":40,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Scripting language; Data quality; Data science; Analytics; Big data; Data mining; Qualitative property; Taxonomy (biology); Human error; Data analysis; Machine learning; Engineering; Reliability engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.05339864,0.002031484,0.002210829,0.007073906,0.004202012,0.009652482,0.005416725,0.002054479,0.01234768],"category_scores_gemma":[0.1973461,0.001184802,0.002950808,0.01112105,0.004445929,0.009192078,0.009725417,0.004343951,0.005604888],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005196976,"about_ca_system_score_gemma":0.01182278,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005870326,"about_ca_topic_score_gemma":0.004911764,"domain_scores_codex":[0.9257687,0.02979504,0.007117928,0.007476767,0.0281829,0.001658687],"domain_scores_gemma":[0.8062847,0.06957751,0.0143249,0.05223441,0.05603293,0.001545472],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0005591808,0.000259716,0.01360473,0.01033017,0.0005347159,0.0005472692,0.01145716,0.009163247,0.01463893,0.2943898,0.09015132,0.5543637],"study_design_scores_gemma":[0.0001092002,0.0003041657,0.006333708,0.004173258,0.0002629432,0.0009726965,0.007900896,0.02172253,0.03034131,0.2558532,0.6717395,0.0002865991],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006329218,0.002408025,0.9529037,0.006619236,0.0009322039,0.002039136,0.007603204,0.003262329,0.01790278],"genre_scores_gemma":[0.09568972,0.004334341,0.8640237,0.004615379,0.0005877364,0.004134416,0.01180175,0.001528911,0.01328402],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9466013,"threshold_uncertainty_score":0.2824024,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4971943852757241,"score_gpt":0.5118316654115125,"score_spread":0.01463728013578841,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}