{"id":"W4386566644","doi":"10.18653/v1/2023.eacl-main.188","title":"How Many and Which Training Points Would Need to be Removed to Flip this Prediction?","year":2023,"lang":"en","type":"article","venue":"","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Army Research Office; Atomic Energy of Canada Limited","keywords":"Robustness (evolution); Computer science; Text categorization; Cardinality (data modeling); Categorization; Training set; Simple (philosophy); Machine learning; Artificial intelligence; Set (abstract data type); Regular polygon; Algorithm; Data mining; Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006160099,0.0009782931,0.001966913,0.0008665348,0.001577048,0.002029294,0.002214211,0.003068149,0.003848025],"category_scores_gemma":[0.03451188,0.0007410735,0.001115361,0.0005885235,0.002052154,0.004651877,0.001354305,0.003144571,0.002653609],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009636364,"about_ca_system_score_gemma":0.001877975,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003194397,"about_ca_topic_score_gemma":0.003192694,"domain_scores_codex":[0.9979323,0.0005971079,0.000114779,0.0008377414,0.0002756885,0.0002424463],"domain_scores_gemma":[0.986775,0.009263489,0.0007554556,0.001794138,0.0009315301,0.0004803381],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002192497,0.0008232668,0.05474485,0.0005801796,0.000275527,0.000720257,0.0006753176,0.243047,0.01286848,0.03012231,0.02675754,0.6271928],"study_design_scores_gemma":[0.0001453049,0.0004961257,0.008182062,0.0001638772,0.0001322314,0.0004928787,0.0006866215,0.8359692,0.009438238,0.1386555,0.005564091,0.00007384496],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4092214,0.0009235045,0.5615349,0.01768001,0.0005257047,0.0002760378,0.001114633,0.001663304,0.007060576],"genre_scores_gemma":[0.8112677,0.0003336573,0.1816617,0.00161737,0.0002717171,0.000168936,0.001250433,0.0002877913,0.003140677],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006160099,"threshold_uncertainty_score":0.03257811,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05024372021815765,"score_gpt":0.2772746563930931,"score_spread":0.2270309361749354,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}