{"id":"W4415311596","doi":"10.48550/arxiv.2506.14126","title":"From Memorization to Parameter Interference: How Overtraining Experts Harms Model Merging","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Statistics Education and Methodologies","field":"Mathematics","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Fonds de recherche du Québec – Nature et technologies; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research; National Science Foundation","keywords":"Pipeline (software); Leverage (statistics); Reuse; Set (abstract data type); Downstream (manufacturing); Boosting (machine learning); Robustness (evolution)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004087677,0.0003633136,0.0005451919,0.0002160722,0.0000890365,0.0001445149,0.0004661156,0.0002959136,0.0001560226],"category_scores_gemma":[0.008962944,0.0003478544,0.0001232574,0.00013373,0.00003987243,0.00007184961,0.0006651383,0.0004212607,0.00001937842],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001422442,"about_ca_system_score_gemma":0.0002630471,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005945844,"about_ca_topic_score_gemma":0.00001237432,"domain_scores_codex":[0.998154,0.0001949732,0.0004333384,0.0006922832,0.0002250835,0.0003003391],"domain_scores_gemma":[0.9965482,0.002087994,0.0002649714,0.0007814209,0.0002013433,0.0001160368],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003650335,0.000877067,0.03632886,0.002471266,0.001929749,0.00003747607,0.3261899,0.01955304,0.01453612,0.07659459,0.4379442,0.08317275],"study_design_scores_gemma":[0.0008006478,0.00006990326,0.003046625,0.002741363,0.0005470772,0.000002158281,0.01660496,0.1417296,0.03100872,0.7982825,0.003269696,0.001896745],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2435362,0.00006844072,0.7497101,0.0014682,0.003299606,0.0003669889,0.0002147414,0.0001995251,0.001136228],"genre_scores_gemma":[0.4179981,0.00005396238,0.5727561,0.0008321356,0.0003279075,0.0002293288,0.0001833112,0.00004042232,0.007578713],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.7216879,"threshold_uncertainty_score":0.9998974,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4280267706046366,"score_gpt":0.4570584216027669,"score_spread":0.02903165099813038,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}