{"id":"W4400702583","doi":"10.52202/079017-3256","title":"Self-Consuming Generative Models with Curated Data Provably Optimize Human Preferences","year":2024,"lang":"en","type":"article","venue":"","topic":"Generative Adversarial Networks and Image Synthesis","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; Université de Montréal; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Generative model; Computer science; Generative grammar; Retraining; Pairwise comparison; Upload; Artificial intelligence; Function (biology); Machine learning; Stability (learning theory); Synthetic data","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0002871756,0.0002276689,0.0002048736,0.00008791383,0.0002561122,0.001143386,0.001322839,0.00005170332,0.00005218137],"category_scores_gemma":[0.000008862331,0.0001464401,0.00002879879,0.0004922612,0.00006015659,0.003163364,0.0005568769,0.0001555166,0.00002834318],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002754718,"about_ca_system_score_gemma":0.0002093201,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009330042,"about_ca_topic_score_gemma":0.00008181627,"domain_scores_codex":[0.9982018,0.0001205958,0.0002200607,0.000892428,0.0002577259,0.0003073801],"domain_scores_gemma":[0.9987318,0.0000918364,0.00004629104,0.0009155894,0.0001251281,0.00008937386],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004243676,0.0004892877,0.00009869807,0.0002615956,0.002027858,0.000328547,0.008879102,0.3137565,0.01070561,0.5200646,0.03961308,0.1037326],"study_design_scores_gemma":[0.0001359067,0.00008411772,0.000006516761,0.00005177587,0.00002983878,0.000009704943,0.00005978267,0.9905244,0.004170831,0.002808735,0.001862346,0.0002560309],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0009404093,0.0005340831,0.9816998,0.0005699598,0.0001973726,0.0002988458,0.00001273331,0.0006017889,0.01514503],"genre_scores_gemma":[0.4511878,0.00003494696,0.5476406,0.0000902305,0.0001168041,0.00002267703,0.00002662127,0.00001263934,0.0008676816],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.6767679,"threshold_uncertainty_score":0.9998935,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06929622210286662,"score_gpt":0.2768607772941121,"score_spread":0.2075645551912455,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}