{"id":"W4408062977","doi":"10.5220/0013261100003905","title":"Zeroth Order Optimization for Pretraining Language Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal; Group for Research in Decision Analysis","funders":"","keywords":"Computer science; Order (exchange); Zeroth law of thermodynamics; Artificial intelligence; Natural language processing; Physics; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001099072,0.001020572,0.0007775942,0.0005460227,0.0004590168,0.001005716,0.001196851,0.001209527,0.004984705],"category_scores_gemma":[0.005450516,0.0006768752,0.0006864148,0.0004995065,0.001280952,0.001932833,0.001374924,0.002698543,0.001585382],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001233094,"about_ca_system_score_gemma":0.002297628,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00593728,"about_ca_topic_score_gemma":0.009514919,"domain_scores_codex":[0.9993413,0.0002263604,0.00004100646,0.0001320352,0.0001623682,0.00009693763],"domain_scores_gemma":[0.9979171,0.001515154,0.00009028239,0.0001826884,0.0002276862,0.0000670948],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002567975,0.0001320376,0.0008652232,0.0002946428,0.00007246041,0.0001054981,0.0002259514,0.7419505,0.01199409,0.0343288,0.005489813,0.2042843],"study_design_scores_gemma":[0.000005624923,0.00001799588,0.00005147059,0.0000075404,0.000004058212,0.000009604502,0.00000849044,0.9914682,0.001973772,0.00596565,0.0004833481,0.000004223525],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02316443,0.0003447982,0.9724208,0.0003762814,0.0000730789,0.0000408194,0.00007722094,0.001431885,0.002070778],"genre_scores_gemma":[0.4280819,0.0004223573,0.5587813,0.0006809334,0.00009631385,0.0002571226,0.0006010252,0.001141572,0.009937478],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00593728,"threshold_uncertainty_score":0.01667553,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01347013928825409,"score_gpt":0.293407060851763,"score_spread":0.279936921563509,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}