{"id":"W4386566845","doi":"10.18653/v1/2023.eacl-main.13","title":"Do we need Label Regularization to Fine-tune Pre-trained Language Models?","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; University of Waterloo","funders":"","keywords":"Pascal (unit); Regularization (linguistics); Computer science; Natural language processing; Artificial intelligence; Theoretical computer science; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000229307,0.0001015955,0.0001079958,0.0001680087,0.00006047552,0.0001386801,0.0005783429,0.00005380845,0.00002781286],"category_scores_gemma":[0.00004147392,0.00009180446,0.00002774973,0.0009033754,0.000006241191,0.0003773163,0.0003195284,0.00006110709,0.0001622271],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002409258,"about_ca_system_score_gemma":0.0000330224,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000590512,"about_ca_topic_score_gemma":0.00003069944,"domain_scores_codex":[0.9989043,0.00002789289,0.0001768364,0.0003652453,0.0002665937,0.0002591077],"domain_scores_gemma":[0.9991565,0.00003530307,0.00002780008,0.0006347513,0.00005035792,0.00009529027],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000008613814,0.00004908613,0.00003956425,0.00003583946,0.00001817919,0.0000317488,0.01485518,0.1387569,0.02272482,0.628408,0.008331218,0.1867409],"study_design_scores_gemma":[0.0002288338,0.00002124602,0.00008034298,0.00001842504,0.000001905875,0.000001762462,0.0001012905,0.9776849,0.001465349,0.0198637,0.0004059181,0.0001263963],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0329826,0.00004033443,0.9514115,0.01025148,0.0001991899,0.0001980654,0.00000153247,0.0008980983,0.004017226],"genre_scores_gemma":[0.7131209,0.00001069007,0.2443586,0.0007156302,0.000138529,0.00003434021,0.00001063092,0.00001980403,0.0415909],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8389279,"threshold_uncertainty_score":0.3743677,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03158016076292302,"score_gpt":0.2743124755449816,"score_spread":0.2427323147820586,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}