{"id":"W4297841867","doi":"10.21437/interspeech.2022-10761","title":"Daft-Exprt: Cross-Speaker Prosody Transfer on Any Text for Expressive Speech Synthesis","year":2022,"lang":"en","type":"article","venue":"Interspeech 2022","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ubisoft (Canada)","funders":"","keywords":"Prosody; Computer science; Speech synthesis; Speech recognition; Transfer (computing); Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009278741,0.001233376,0.0005582059,0.0002489874,0.0002666481,0.0006383726,0.001481478,0.0009852932,0.008162303],"category_scores_gemma":[0.002034144,0.0003246837,0.0009129012,0.0001227344,0.000462859,0.001068024,0.00211043,0.001949543,0.003424792],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002977506,"about_ca_system_score_gemma":0.0004375832,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000951498,"about_ca_topic_score_gemma":0.001431298,"domain_scores_codex":[0.9996499,0.00008412806,0.00001723451,0.0001222784,0.00009508589,0.00003131744],"domain_scores_gemma":[0.9996015,0.0002002753,0.00001910781,0.00009710409,0.00005237169,0.00002961468],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006150588,0.0002949603,0.0006591768,0.0003668607,0.0002239098,0.0005883347,0.000287843,0.3955076,0.09570426,0.008225786,0.01205413,0.4854721],"study_design_scores_gemma":[0.00004545154,0.0002125647,0.0001829201,0.00002293884,0.00002784206,0.000199974,0.00002253214,0.9639949,0.02413775,0.00497185,0.006154557,0.00002666806],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02470925,0.0006491448,0.9560316,0.0002326616,0.000404907,0.0001904366,0.0005085266,0.01143992,0.005833603],"genre_scores_gemma":[0.5836082,0.0006115011,0.3837751,0.0006375303,0.0002122818,0.0006794605,0.002953858,0.002201454,0.0253207],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008162303,"threshold_uncertainty_score":0.0273056,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02503636111819494,"score_gpt":0.2806554597688282,"score_spread":0.2556190986506332,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}