{"id":"W6929137473","doi":"10.48448/9jnz-7f73","title":"The Art of Abstention: Selective Prediction and Error Regularization for Natural Language Processing","year":2021,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Regularization (linguistics); Computation; Classifier (UML); Confidence interval; Natural language; Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007977723,0.0001725612,0.0001879254,0.0003194968,0.0004174,0.0001644277,0.0002464173,0.0001185741,0.00003036831],"category_scores_gemma":[0.0003977182,0.0001265718,0.00003875403,0.001114472,0.0009949524,0.0001984367,0.00007448615,0.0001581422,0.000006253665],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001378915,"about_ca_system_score_gemma":0.0005283854,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001771713,"about_ca_topic_score_gemma":0.0004663628,"domain_scores_codex":[0.9983848,0.00004082021,0.0002505822,0.0004785723,0.0005793106,0.000265899],"domain_scores_gemma":[0.9986904,0.00005352917,0.0004586246,0.0002452407,0.0005047605,0.00004750426],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003326774,0.0005588926,0.001159951,0.002595215,0.000437742,0.00001083025,0.006568037,0.0004055658,0.3828768,0.01270422,0.3250306,0.2673194],"study_design_scores_gemma":[0.003162039,0.0005361941,0.003860692,0.003509596,0.0005170228,0.0001523749,0.009998843,0.7887022,0.02006892,0.002235518,0.1659347,0.001321911],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.006711419,0.1122266,0.1441074,0.002907415,0.009528833,0.01700055,0.00256352,0.003001806,0.7019525],"genre_scores_gemma":[0.2757411,0.00006690566,0.02060996,0.00005225628,0.0007986471,0.0001286174,0.001095778,0.0005398187,0.7009669],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.7882967,"threshold_uncertainty_score":0.5161448,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01382675952895413,"score_gpt":0.2992396827642083,"score_spread":0.2854129232352541,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}