{"id":"W3015276915","doi":"10.1145/3318464.3389696","title":"Complaint-driven Training Data Debugging for Query 2.0","year":2020,"lang":"en","type":"article","venue":"","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"Innovative Research Group Project of the National Natural Science Foundation of China","keywords":"Debugging; Retraining; Heuristic; Set (abstract data type); Training set; Inference; Training (meteorology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01289937,0.001871469,0.001402588,0.001279031,0.0009776615,0.001914895,0.005793791,0.002267597,0.002634885],"category_scores_gemma":[0.0533774,0.001039614,0.001053398,0.001050562,0.001688235,0.004795515,0.003782389,0.003695033,0.001561059],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001375258,"about_ca_system_score_gemma":0.002885,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004771032,"about_ca_topic_score_gemma":0.008450376,"domain_scores_codex":[0.9903913,0.004480923,0.0006394692,0.001777258,0.002214619,0.0004964659],"domain_scores_gemma":[0.9603453,0.02192564,0.002241771,0.01087288,0.003756385,0.0008579997],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00241521,0.002030407,0.03643968,0.0008425361,0.0003597375,0.001039723,0.002605955,0.1665228,0.04728274,0.01399188,0.09209155,0.6343778],"study_design_scores_gemma":[0.0001164975,0.0002376221,0.001251972,0.00002547275,0.00002784148,0.0002322679,0.0001719887,0.9690905,0.01826748,0.004877011,0.005656744,0.00004454438],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07133205,0.0006776018,0.7313036,0.001667986,0.0001808597,0.0004188863,0.0006783616,0.1912585,0.002482147],"genre_scores_gemma":[0.4931446,0.0001688931,0.4960791,0.001346037,0.0001194338,0.000280404,0.002198792,0.004461572,0.002201092],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01289937,"threshold_uncertainty_score":0.06821918,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3264365626267824,"score_gpt":0.3428036694085752,"score_spread":0.01636710678179271,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}