{"id":"W6929071316","doi":"10.48448/a98m-1a69","title":"End-to-End Self-Debiasing Framework for Robust NLU Training","year":2021,"lang":"en","type":"other","venue":"Open MIND","topic":"Chemical and Physical Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; Université de Montréal","funders":"","keywords":"Debiasing; Natural language understanding; Training (meteorology); Simple (philosophy); Training set; Task (project management)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0000651888,0.0002277377,0.0003437404,0.00001521526,0.00006613777,0.00008280071,0.0003332251,0.0003241786,0.003024559],"category_scores_gemma":[0.000107347,0.0002085554,0.0001339834,0.00007185352,0.00003459966,8.933155e-7,0.0004113977,0.0001279046,0.00003604599],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001109416,"about_ca_system_score_gemma":0.00006533216,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002843985,"about_ca_topic_score_gemma":0.00008181229,"domain_scores_codex":[0.9989345,0.00001806552,0.0001403271,0.0005514954,0.00008653678,0.0002690603],"domain_scores_gemma":[0.9994767,0.00004309361,0.00007781357,0.0002701953,0.0000291178,0.0001030843],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001973815,0.0003742566,0.00004380702,0.0002030581,0.001567969,0.00001827533,0.001021007,0.00002044912,0.1247704,0.0002103546,0.3481178,0.5234553],"study_design_scores_gemma":[0.0002023341,0.00006906199,0.000005766061,0.0002231029,0.00005091173,0.000001940901,0.00009466129,0.00000496715,0.02875517,0.00009433781,0.9701928,0.0003048928],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.003179857,0.002955419,0.0112504,0.0006796839,0.0005569453,0.001222984,0.0005265372,0.000008569596,0.9796196],"genre_scores_gemma":[0.005328588,0.0002452906,0.4267933,0.0006640406,0.00411965,0.0001428737,0.001395686,0.0004218374,0.5608887],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.6220751,"threshold_uncertainty_score":0.9978868,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0669392359194059,"score_gpt":0.3259969365856715,"score_spread":0.2590577006662657,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}