{"id":"W4415540898","doi":"10.1145/3746027.3754745","title":"Pseudo-Autoregressive Neural Codec Language Models for Efficient Zero-Shot Text-to-Speech Synthesis","year":2025,"lang":"","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Microsoft (Canada)","funders":"National Natural Science Foundation of China","keywords":"Autoregressive model; Language model; Set (abstract data type); Inference; Codec; Face (sociological concept); Speech processing; Speech synthesis","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.001018353,0.0008558608,0.001065149,0.001046273,0.0007447934,0.001115528,0.002445301,0.0004060846,0.0009004321],"category_scores_gemma":[0.001289077,0.0007726532,0.0007720948,0.001504583,0.0002090527,0.0004482298,0.0008643898,0.0003508771,0.0005251709],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002986509,"about_ca_system_score_gemma":0.0004831064,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000190672,"about_ca_topic_score_gemma":0.00007200846,"domain_scores_codex":[0.994107,0.0003356522,0.00115922,0.002015635,0.0009149436,0.001467533],"domain_scores_gemma":[0.9938671,0.002747985,0.0002978972,0.001722472,0.0007414437,0.0006230983],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002108107,0.0006940097,0.00001281655,0.0001860815,0.0002317404,0.00009836427,0.002167668,0.006734187,0.002678426,0.02597159,0.01591709,0.9450972],"study_design_scores_gemma":[0.0007231832,0.0001292943,0.00007774804,0.0004621225,0.0001750731,0.00003602392,0.0006817795,0.8764149,0.1173096,0.001768534,0.001414165,0.0008075829],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03367071,0.0005371466,0.9050375,0.007644249,0.002360189,0.002407355,0.000240033,0.0005050578,0.04759775],"genre_scores_gemma":[0.8394675,0.00003788143,0.1344981,0.005554986,0.0001662506,0.0006902854,0.00000756563,0.00006465961,0.01951277],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9442896,"threshold_uncertainty_score":0.9999214,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03375509139618933,"score_gpt":0.2939912735728801,"score_spread":0.2602361821766908,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}