{"id":"W2984256198","doi":"10.18653/v1/d19-6115","title":"Unlearn Dataset Bias in Natural Language Inference by Fitting the Residual","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":165,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University","keywords":"Debiasing; Residual; Computer science; Artificial intelligence; Inference; Machine learning; Benchmark (surveying); Natural language processing; Baseline (sea); Algorithm; Psychology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01288737,0.001675801,0.001366721,0.001574899,0.0007991167,0.001926578,0.003480345,0.0024144,0.001890613],"category_scores_gemma":[0.04448891,0.0009059005,0.001561196,0.00135176,0.002449591,0.005383181,0.003810323,0.005803228,0.001563913],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001836587,"about_ca_system_score_gemma":0.002100905,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00516036,"about_ca_topic_score_gemma":0.008718845,"domain_scores_codex":[0.9953675,0.001939424,0.0002995525,0.001504584,0.0006241307,0.000264734],"domain_scores_gemma":[0.9790446,0.01218867,0.001338254,0.005525681,0.001583331,0.0003195026],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009166778,0.0004361309,0.03897644,0.0005912593,0.0005151083,0.0003453958,0.001046535,0.4591457,0.01562644,0.02534315,0.01859988,0.4384573],"study_design_scores_gemma":[0.00005637603,0.0001307543,0.001489622,0.00005887319,0.00003853979,0.0001314721,0.00008362789,0.9633226,0.005550753,0.02668572,0.00241607,0.00003553544],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09293655,0.001346562,0.890214,0.001591764,0.000138663,0.0002177641,0.0009557985,0.01065667,0.001942219],"genre_scores_gemma":[0.6014112,0.0003426177,0.3873802,0.001963649,0.0001537203,0.0004185831,0.004862491,0.000994792,0.002472752],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01288737,"threshold_uncertainty_score":0.06815577,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01741412908958099,"score_gpt":0.3055379553608717,"score_spread":0.2881238262712907,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}