{"id":"W4372259784","doi":"10.1109/icassp49357.2023.10095515","title":"Grad-StyleSpeech: Any-Speaker Adaptive Text-to-Speech Synthesis with Diffusion Models","year":2023,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Speech recognition; Speech synthesis; Similarity (geometry); Speech processing; Generative grammar; Generative model; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000791225,0.001451439,0.0007203082,0.0005242654,0.000286363,0.0007786019,0.001292499,0.0009975967,0.01429641],"category_scores_gemma":[0.00184363,0.0003276105,0.0007277438,0.0003813204,0.0003813955,0.001053098,0.001805196,0.001483011,0.007506331],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003547556,"about_ca_system_score_gemma":0.0006394553,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002741517,"about_ca_topic_score_gemma":0.005222023,"domain_scores_codex":[0.9995271,0.00007991023,0.00002764733,0.0001558821,0.0001698514,0.00003964375],"domain_scores_gemma":[0.9996293,0.0001495499,0.00001821349,0.00009153414,0.00007357038,0.00003789413],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0009589407,0.0002740231,0.0007987835,0.0004801208,0.0001848157,0.0004145078,0.0002528742,0.1082069,0.1115268,0.009868595,0.04411468,0.722919],"study_design_scores_gemma":[0.0002388007,0.0003386545,0.0006387056,0.00004643176,0.00005204074,0.0004328538,0.00006903799,0.8594557,0.07284107,0.01287051,0.0529285,0.00008782575],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01467829,0.001338169,0.9220688,0.0002964285,0.0005159146,0.0001727465,0.001958803,0.04969534,0.009275545],"genre_scores_gemma":[0.3101025,0.001042258,0.6418599,0.0006261767,0.0002740974,0.000522458,0.0112151,0.007773211,0.02658428],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01429641,"threshold_uncertainty_score":0.04782629,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04366180071214854,"score_gpt":0.2385083710509253,"score_spread":0.1948465703387767,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}