{"id":"W4393147067","doi":"10.1609/aaai.v38i16.29747","title":"UniCATS: A Unified Context-Aware Text-to-Speech Framework with Contextual VQ-Diffusion and Vocoding","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Context (archaeology); Computer science; Linguistics; Psychology; Speech recognition; Natural language processing; Biology; Paleontology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004602406,0.001356669,0.000806117,0.0005495722,0.000359575,0.0007672657,0.001246573,0.0007374426,0.003228254],"category_scores_gemma":[0.001015211,0.0003395337,0.0009121773,0.0004037199,0.0005042045,0.001109933,0.001442497,0.001203728,0.001568318],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004361396,"about_ca_system_score_gemma":0.00119361,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008996177,"about_ca_topic_score_gemma":0.01402648,"domain_scores_codex":[0.9995911,0.00007626681,0.00002538642,0.0001243552,0.0001348799,0.00004803528],"domain_scores_gemma":[0.9997291,0.00006657781,0.00002313535,0.00004391583,0.0001053359,0.00003191987],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006773459,0.0001508,0.0007403811,0.0003144523,0.0001272486,0.0003885046,0.0002290132,0.1597493,0.1222908,0.0120631,0.009542239,0.693727],"study_design_scores_gemma":[0.00004196372,0.0001937309,0.0003242437,0.00002173404,0.00004191704,0.0001701822,0.00006219266,0.9513777,0.03206662,0.004409914,0.0112473,0.00004246602],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.009660997,0.0009746504,0.9831679,0.0000983269,0.0001801695,0.00008727029,0.0002696587,0.004346986,0.001214015],"genre_scores_gemma":[0.3248662,0.001331736,0.658631,0.0004507861,0.0004182476,0.0003725688,0.00243733,0.0009263879,0.01056568],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008996177,"threshold_uncertainty_score":0.01788765,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06781807521491127,"score_gpt":0.2937481810081966,"score_spread":0.2259301057932853,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}