{"id":"W4404305672","doi":"10.48550/arxiv.2407.18213","title":"Scaling Trends in Language Model Robustness","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Robustness (evolution); Computer science; Scale (ratio); Language model; Natural language processing; Geography; Cartography; Chemistry","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006270621,0.001131946,0.0008013289,0.001469926,0.0006787362,0.002373905,0.001403121,0.001780941,0.003613234],"category_scores_gemma":[0.0514122,0.0006999159,0.001205914,0.0008866624,0.002807415,0.009891029,0.002707038,0.006550228,0.001234838],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001374367,"about_ca_system_score_gemma":0.0007549278,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002107731,"about_ca_topic_score_gemma":0.001540223,"domain_scores_codex":[0.9961671,0.001153519,0.0001805626,0.001066357,0.001047708,0.0003847669],"domain_scores_gemma":[0.971522,0.0167997,0.001495568,0.008031308,0.001558168,0.0005932536],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007546115,0.0004472518,0.02506712,0.0007681304,0.0004922937,0.0006127274,0.001663209,0.4286068,0.02897525,0.2301569,0.03207216,0.2503836],"study_design_scores_gemma":[0.00004823359,0.0004275153,0.006033821,0.0001705264,0.00009574317,0.0005366146,0.0004733934,0.7829982,0.01901182,0.1700181,0.02008291,0.0001031043],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4990205,0.01325893,0.4079153,0.02895666,0.001202954,0.0002346179,0.002192435,0.008237969,0.03898069],"genre_scores_gemma":[0.9587035,0.001970696,0.03389074,0.001132293,0.0003975919,0.0000819533,0.0008673177,0.0005337776,0.002422082],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006270621,"threshold_uncertainty_score":0.03316259,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05449121822374094,"score_gpt":0.2240343428438835,"score_spread":0.1695431246201426,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}