{"id":"W4415991972","doi":"10.48550/arxiv.2508.21631","title":"Towards Improved Speech Recognition through Optimized Synthetic Data Generation","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Synthetic data; Training set; Voice activity detection; Speech synthesis; Acoustic model; Encoding (memory); Pattern recognition (psychology)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0009146706,0.0005078237,0.0006237942,0.0002345426,0.0002267872,0.0005353197,0.00320454,0.0005419724,0.0004745562],"category_scores_gemma":[0.0008928782,0.0005075318,0.0002352941,0.0003975327,0.00007396509,0.001102625,0.003839551,0.0007102528,0.0005277586],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001537658,"about_ca_system_score_gemma":0.0005655616,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003409466,"about_ca_topic_score_gemma":0.00005005637,"domain_scores_codex":[0.9960073,0.0003635433,0.0007430188,0.001996439,0.0004365225,0.0004532251],"domain_scores_gemma":[0.9950252,0.0001897279,0.0003981336,0.003836134,0.000430104,0.0001207249],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003284299,0.000280724,0.0001440152,0.0001844339,0.0002629079,0.00004723616,0.0002933869,0.0000322116,0.003124446,0.0001454324,0.008297286,0.9871551],"study_design_scores_gemma":[0.001704804,0.00009164454,0.0009719884,0.0008479251,0.0004371101,0.00007052635,0.00007979333,0.7465713,0.2250714,0.01047047,0.01167445,0.002008668],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05938096,0.0004704013,0.9000242,0.004978691,0.00618733,0.001457399,0.0006972426,0.001088424,0.02571538],"genre_scores_gemma":[0.05224708,0.001490269,0.9378926,0.002457901,0.001062319,0.0002918455,0.002639347,0.00005347183,0.001865166],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9851464,"threshold_uncertainty_score":0.9997376,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1907698283837188,"score_gpt":0.3255550460538047,"score_spread":0.1347852176700859,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}