{"id":"W4281400028","doi":"10.1145/3533028.3533305","title":"How I stopped worrying about training data bugs and started complaining","year":2022,"lang":"en","type":"article","venue":"","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"Amazon Web Services; Google; National Science Foundation","keywords":"Debugging; Computer science; Downstream (manufacturing); Complaint; Inference; Training (meteorology); Training set; Set (abstract data type); Quality (philosophy); Data quality; Data integrity; Data set; Data science; Machine learning; Artificial intelligence; Software engineering; Computer security; Engineering; Programming language; Operations management","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02575825,0.001566065,0.001040308,0.001370296,0.003781141,0.008187251,0.003814979,0.007886472,0.01786449],"category_scores_gemma":[0.1743801,0.001349147,0.001481344,0.001044142,0.006486671,0.0157392,0.003980239,0.01298297,0.01591504],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001904565,"about_ca_system_score_gemma":0.003017023,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004025058,"about_ca_topic_score_gemma":0.003614845,"domain_scores_codex":[0.985615,0.006209565,0.000629183,0.003018223,0.003492556,0.001035435],"domain_scores_gemma":[0.940159,0.02432649,0.004912641,0.01223685,0.01411171,0.004253375],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0009537676,0.0004731602,0.01724657,0.0007064335,0.0003427144,0.001811639,0.01184585,0.006949018,0.007060584,0.04978564,0.5399017,0.3629229],"study_design_scores_gemma":[0.0002516931,0.0009101211,0.006731272,0.002369123,0.0003046858,0.00721374,0.009928205,0.03243523,0.01825448,0.208746,0.7118593,0.0009959984],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.03344676,0.006257821,0.3797237,0.5078315,0.01883247,0.0005467113,0.001679318,0.01451209,0.03716972],"genre_scores_gemma":[0.3567652,0.00519811,0.3007171,0.234512,0.006581986,0.0008138162,0.001998043,0.009876973,0.08353682],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.02575825,"threshold_uncertainty_score":0.1362243,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09801001456079682,"score_gpt":0.283168824638234,"score_spread":0.1851588100774372,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}