{"id":"W4416251585","doi":"10.1109/ijcnn64981.2025.11228044","title":"Training Dynamics of a 1.7B LLaMa Model: A Data-Efficient Approach","year":2025,"lang":"","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Training (meteorology); Dynamics (music); Stability (learning theory); Sample (material); Training set; Qualitative property; Qualitative research; Variety (cybernetics)","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003243464,0.001538099,0.001216236,0.0007161634,0.0009533482,0.002699203,0.003585576,0.00190443,0.007330558],"category_scores_gemma":[0.02225688,0.001460486,0.001281395,0.0006250732,0.001263987,0.005378929,0.002945437,0.005505482,0.004749338],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001947325,"about_ca_system_score_gemma":0.002846516,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01317127,"about_ca_topic_score_gemma":0.02562619,"domain_scores_codex":[0.9985067,0.0005760597,0.00008242807,0.0004704446,0.0002295195,0.0001348544],"domain_scores_gemma":[0.9937363,0.003970453,0.0001351335,0.001169512,0.0007759326,0.0002126328],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00128453,0.0004515274,0.007612625,0.0004614625,0.0002471091,0.0003643465,0.001589437,0.7105737,0.01451092,0.01361572,0.03212371,0.2171649],"study_design_scores_gemma":[0.0000381172,0.00006015734,0.0002291023,0.00002804027,0.00001973179,0.00003134748,0.0001175504,0.98512,0.00259431,0.008337867,0.00340725,0.0000164117],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.165659,0.001399155,0.7582539,0.005040733,0.0004551474,0.0004102213,0.003547376,0.05591439,0.00932011],"genre_scores_gemma":[0.6107469,0.0003665637,0.3672367,0.001690546,0.000130795,0.0009887523,0.006197364,0.005717831,0.006924662],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01317127,"threshold_uncertainty_score":0.02618921,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1168107725970234,"score_gpt":0.3068613338009837,"score_spread":0.1900505612039602,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}