{"id":"W2548122763","doi":"10.14778/2994509.2994514","title":"ActiveClean","year":2016,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":244,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"MNIST database; Computer science; Context (archaeology); Support vector machine; Data mining; Convergence (economics); Process (computing); Class (philosophy); Iterative and incremental development; Machine learning; Artificial intelligence; Deep learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005524416,0.003281085,0.002910606,0.003398271,0.002098419,0.007996758,0.007369641,0.003558987,0.04162381],"category_scores_gemma":[0.02294586,0.002480949,0.003860537,0.003341832,0.001404523,0.007642557,0.007066833,0.00456045,0.03641694],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008763244,"about_ca_system_score_gemma":0.003094282,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003690959,"about_ca_topic_score_gemma":0.007741269,"domain_scores_codex":[0.9951501,0.001209463,0.0004006198,0.001279485,0.001660123,0.0003001515],"domain_scores_gemma":[0.990415,0.004144664,0.0003539688,0.003346487,0.0014871,0.0002528988],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005766489,0.0002924403,0.004378524,0.001855119,0.0006870286,0.0004468172,0.0008711849,0.05785158,0.006938002,0.04163717,0.4085335,0.475932],"study_design_scores_gemma":[0.0001804244,0.0001143024,0.0008681212,0.0002539227,0.0001296521,0.0006816341,0.0004368314,0.435555,0.01938186,0.1141362,0.4281086,0.0001533892],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00346497,0.001361244,0.8787732,0.001261081,0.000674281,0.0003527406,0.006805339,0.09451578,0.01279131],"genre_scores_gemma":[0.06609979,0.002148384,0.8232331,0.002652606,0.0003656152,0.001181668,0.04470693,0.03090976,0.02870207],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.04162381,"threshold_uncertainty_score":0.1392455,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009115700324434017,"score_gpt":0.2121947867126384,"score_spread":0.2030790863882044,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}