{"id":"W3162841994","doi":"10.18653/v1/2021.findings-acl.312","title":"Learning Robust Latent Representations for Controllable Speech Synthesis","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vanguard College","funders":"","keywords":"Computer science; Transformer; Speech recognition; Latent variable; Artificial intelligence; Encoder; Probabilistic latent semantic analysis; Mutual information; Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.000625525,0.0002869753,0.0005379784,0.0002156124,0.0002616495,0.00112203,0.0009049176,0.0002735418,0.0009713783],"category_scores_gemma":[0.001580193,0.0002787761,0.0004607849,0.0002474604,0.00002984888,0.0002466669,0.0008642804,0.00039792,0.00009923424],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000877545,"about_ca_system_score_gemma":0.0002695514,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000185207,"about_ca_topic_score_gemma":0.00007326787,"domain_scores_codex":[0.9975334,0.0002202603,0.0004754738,0.001011463,0.0003693585,0.0003899825],"domain_scores_gemma":[0.9967094,0.001422383,0.0002382668,0.0009278657,0.0005552848,0.0001468047],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004867062,0.0006137789,0.001285542,0.0005135544,0.001166287,0.0002254724,0.001052215,0.0341992,0.002103881,0.01059464,0.01977671,0.9284201],"study_design_scores_gemma":[0.0008537098,0.00006291829,0.0006749979,0.0003987931,0.0002296746,0.00007011992,0.0009233864,0.9032774,0.07968661,0.004925396,0.007685137,0.001211832],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.004118729,0.0001657882,0.9546327,0.004031605,0.001071809,0.0008433282,0.00001883425,0.0006388694,0.03447833],"genre_scores_gemma":[0.07087207,0.0002183266,0.9103381,0.0004756908,0.0002200911,0.0009794009,0.00006693334,0.00004329049,0.0167861],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9272082,"threshold_uncertainty_score":0.9999664,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05897369964401934,"score_gpt":0.2807863436266693,"score_spread":0.2218126439826499,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}