{"id":"W2963912924","doi":"","title":"Sample-efficient adaptive text-to-speech","year":2018,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Naturalness; Speech recognition; Artificial neural network; Embedding; Benchmark (surveying); Similarity (geometry); Stochastic gradient descent; Speaker recognition; Mean opinion score; Gradient descent; Artificial intelligence; Sample (material); Speaker diarisation; Metric (unit)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000937011,0.001378621,0.001045036,0.0005616763,0.0003308087,0.0006466391,0.002420613,0.00118455,0.004673542],"category_scores_gemma":[0.003682818,0.0005184574,0.0008202239,0.0005381174,0.0006024945,0.001941316,0.001843533,0.001611073,0.003623642],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005489638,"about_ca_system_score_gemma":0.0006855708,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002257338,"about_ca_topic_score_gemma":0.004104915,"domain_scores_codex":[0.9993085,0.0001405385,0.0000437245,0.000269048,0.00017978,0.00005851192],"domain_scores_gemma":[0.9986432,0.0005613267,0.00008035917,0.0004085659,0.0002340172,0.00007247877],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008822295,0.0003730831,0.0009633872,0.0002029048,0.000174828,0.000264578,0.0001813932,0.2659616,0.09119868,0.004289791,0.006145164,0.6293624],"study_design_scores_gemma":[0.00002762325,0.0001044254,0.0002211362,0.000006940112,0.00002187504,0.0000979193,0.00002221077,0.9719623,0.02244444,0.003309593,0.00176709,0.00001451457],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02219785,0.0004645149,0.9673613,0.0001418863,0.0001590298,0.00008202593,0.0002970945,0.00747208,0.00182427],"genre_scores_gemma":[0.5408862,0.0002761032,0.4465953,0.0003436939,0.0001966388,0.0003166809,0.001772865,0.001014463,0.008598028],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004673542,"threshold_uncertainty_score":0.01563454,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08424185677777325,"score_gpt":0.1921571094293518,"score_spread":0.1079152526515785,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}