{"id":"W6891677426","doi":"10.48448/34dm-mp79","title":"Do we need Label Regularization to Fine-tune Pre-trained Language Models?","year":2023,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; University of Waterloo","funders":"","keywords":"Regularization (linguistics); Artificial neural network; Language model; Training set; Set (abstract data type); Deep learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00148909,0.0006505075,0.0005933136,0.002901232,0.0002693091,0.0004066723,0.001848018,0.0004037622,0.0008166066],"category_scores_gemma":[0.0007053784,0.0006138568,0.00008190317,0.007048622,0.0007414639,0.0005481886,0.0005874153,0.0003942857,0.006618697],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005318095,"about_ca_system_score_gemma":0.0008142719,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000879865,"about_ca_topic_score_gemma":0.005223628,"domain_scores_codex":[0.994583,0.00009070729,0.0005470936,0.001575385,0.002071886,0.001131923],"domain_scores_gemma":[0.9971173,0.00008555524,0.0004126851,0.001604973,0.0002852085,0.0004942109],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009801787,0.000517856,0.00003273917,0.0002198405,0.0001339269,0.0001167867,0.005930557,0.01764107,0.09280621,0.05016107,0.7838313,0.04851063],"study_design_scores_gemma":[0.004515794,0.0006453827,0.0001442047,0.003367245,0.0003242294,0.00004489098,0.00261381,0.7613817,0.003429189,0.01961713,0.1992751,0.004641238],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.002305002,0.003081094,0.1171698,0.01072075,0.005443346,0.009213154,0.004460289,0.02741616,0.8201904],"genre_scores_gemma":[0.01016874,0.00006838436,0.03269216,0.0002871366,0.0009096274,0.0001185793,0.0003166567,0.003346687,0.9520921],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.7437407,"threshold_uncertainty_score":0.9996313,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03342531541763453,"score_gpt":0.3177216516896725,"score_spread":0.284296336272038,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}