{"id":"W7123353892","doi":"10.1109/cdc57313.2025.11312306","title":"Preventing Model Collapse when Training LLMs with Synthetic Data","year":2025,"lang":"","type":"article","venue":"","topic":"Generative Adversarial Networks and Image Synthesis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Synthetic data; Training (meteorology); Generative model; Simple (philosophy); Training set; Quality (philosophy)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.001397895,0.0005134968,0.0006354838,0.0002011488,0.0007504193,0.001362508,0.003637518,0.0001459196,0.000269319],"category_scores_gemma":[0.0002596988,0.0004243508,0.0001072365,0.0009149995,0.0002491015,0.002136286,0.002734785,0.0003248938,0.00002923141],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000690251,"about_ca_system_score_gemma":0.001261495,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001024444,"about_ca_topic_score_gemma":0.0002307928,"domain_scores_codex":[0.9956697,0.0003172836,0.0006908921,0.001895126,0.0004940728,0.00093292],"domain_scores_gemma":[0.9956983,0.0003391618,0.0002491575,0.003242332,0.0002731071,0.0001979857],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009139982,0.0003417643,0.00008939533,0.0001427405,0.0006937984,0.00003889463,0.005734694,0.3568959,0.0008175678,0.01901128,0.03299034,0.5831522],"study_design_scores_gemma":[0.0006133433,0.00008534581,0.00001959095,0.0006488878,0.000217419,0.000008183308,0.0005396673,0.9899763,0.0009311127,0.003144953,0.003314624,0.0005005781],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0003425849,0.0006805003,0.927873,0.003476236,0.0006154887,0.000568737,0.00002835581,0.0001091003,0.06630602],"genre_scores_gemma":[0.6252456,0.00006933564,0.3511788,0.0007130782,0.0001227123,0.0000161117,0.000008498212,0.00002224712,0.02262362],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.6330804,"threshold_uncertainty_score":0.9998208,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06272020910394469,"score_gpt":0.2721604772171564,"score_spread":0.2094402681132117,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}