{"id":"W6929137473","doi":"10.48448/9jnz-7f73","title":"The Art of Abstention: Selective Prediction and Error Regularization for Natural Language Processing","year":2021,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Regularization (linguistics); Computation; Classifier (UML); Confidence interval; Natural language; Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008165948,0.001172716,0.001305711,0.0009423353,0.001023766,0.002870507,0.002827478,0.002348115,0.005357497],"category_scores_gemma":[0.04326754,0.0009154358,0.001045665,0.001009037,0.003204074,0.007076921,0.004169714,0.006724667,0.002291886],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001511051,"about_ca_system_score_gemma":0.001593783,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003300058,"about_ca_topic_score_gemma":0.004648359,"domain_scores_codex":[0.9954911,0.002201971,0.0001830153,0.0009472611,0.0009575894,0.0002191351],"domain_scores_gemma":[0.9735483,0.01906272,0.0008254173,0.004862059,0.001312855,0.0003887042],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005941491,0.0002446022,0.004369277,0.0004231831,0.0002362696,0.000245844,0.0005798266,0.2468566,0.006259667,0.2692003,0.03200335,0.4389869],"study_design_scores_gemma":[0.00002139979,0.00003441298,0.0002241741,0.00003890001,0.00001626088,0.00005339445,0.00002052062,0.8141459,0.001652153,0.1808855,0.002890451,0.00001703833],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01266465,0.001393029,0.9754378,0.003759688,0.0001430543,0.00004452059,0.0001684582,0.00190641,0.004482416],"genre_scores_gemma":[0.6071581,0.00160977,0.3754038,0.002098162,0.0007214915,0.0002927129,0.0006241345,0.00160384,0.01048797],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008165948,"threshold_uncertainty_score":0.04318613,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01382675952895413,"score_gpt":0.2992396827642083,"score_spread":0.2854129232352541,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}