{"id":"W4399527282","doi":"10.1109/taffc.2024.3412152","title":"Controllable Multi-Speaker Emotional Speech Synthesis With an Emotion Representation of High Generalization Capability","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Affective Computing","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Windsor","funders":"Natural Science Foundation of Anhui Province; National Natural Science Foundation of China","keywords":"Generalization; Speech recognition; Emotion recognition; Representation (politics); Speaker recognition; Computer science; Emotion classification; Psychology; Artificial intelligence; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002670975,0.0004551438,0.0002639053,0.0001210668,0.0001094182,0.0002608842,0.0003279587,0.0002484937,0.001756237],"category_scores_gemma":[0.0004586031,0.0001296015,0.0004741473,0.00008422615,0.0002059783,0.0003645676,0.0005184801,0.0004697411,0.0005529528],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001207864,"about_ca_system_score_gemma":0.0001222121,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003149432,"about_ca_topic_score_gemma":0.0004133358,"domain_scores_codex":[0.999815,0.00003807779,0.0000113054,0.00006819562,0.00005298302,0.00001435439],"domain_scores_gemma":[0.9998596,0.00005225832,0.00001669343,0.00002794381,0.00003422279,0.000009269274],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002492101,0.00006928647,0.000284676,0.0001188559,0.00004420986,0.0001305168,0.0001591922,0.03868915,0.7444557,0.003544873,0.0007796222,0.2114748],"study_design_scores_gemma":[0.00002963876,0.0002316726,0.0009400129,0.00001143273,0.0000433928,0.0002082088,0.00003838988,0.7492401,0.2422705,0.002227192,0.004733151,0.00002637887],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02845183,0.0001201874,0.969062,0.00004606277,0.00004185848,0.00003282913,0.00003950447,0.0007626686,0.001443064],"genre_scores_gemma":[0.5592787,0.0001961313,0.4351215,0.0001304269,0.00006751691,0.0001717627,0.0002588511,0.0002846151,0.004490432],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.001756237,"threshold_uncertainty_score":0.00587523,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02894511557105801,"score_gpt":0.2795308599183586,"score_spread":0.2505857443473006,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}