{"id":"W4392828353","doi":"10.48550/arxiv.2403.08763","title":"Simple and Scalable Strategies to Continually Pre-train Large Language Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Oak Ridge National Laboratory; Natural Sciences and Engineering Research Council of Canada; Office of Science; Fonds de recherche du Québec – Nature et technologies; Université de Montréal; Deutscher Akademischer Austauschdienst; Canadian Institute for Advanced Research; Canada Excellence Research Chairs, Government of Canada; U.S. Department of Energy","keywords":"Simple (philosophy); Computer science; Scalability; Epistemology; Philosophy","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002944566,0.003012313,0.001394029,0.0009560799,0.0007960837,0.001606141,0.005669427,0.001521165,0.00793799],"category_scores_gemma":[0.01599132,0.001547162,0.001205347,0.0009263566,0.001134203,0.006653386,0.003808687,0.005076426,0.01184273],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001145755,"about_ca_system_score_gemma":0.001996234,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008274904,"about_ca_topic_score_gemma":0.01863209,"domain_scores_codex":[0.9983075,0.000417679,0.0001378592,0.00069266,0.0003171839,0.0001272121],"domain_scores_gemma":[0.9954541,0.001351414,0.0002355139,0.002077959,0.0006800156,0.0002009906],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008190891,0.001045421,0.007057068,0.0005626528,0.0003948497,0.0003096318,0.0006335125,0.1820272,0.03817949,0.007249309,0.03781884,0.723903],"study_design_scores_gemma":[0.0002171221,0.0003020695,0.001586607,0.00005800895,0.00009335944,0.000210517,0.0002163897,0.9456108,0.02597403,0.01136432,0.01425526,0.000111494],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.08293797,0.001703535,0.8352253,0.00126799,0.0005749924,0.0007578361,0.001445697,0.0689408,0.007145828],"genre_scores_gemma":[0.4876392,0.0006899245,0.4902454,0.001162121,0.0002190911,0.001300713,0.005660597,0.00428836,0.008794479],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008274904,"threshold_uncertainty_score":0.02655524,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0430427905600358,"score_gpt":0.2101212016528687,"score_spread":0.1670784110928329,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}