{"id":"W4378231186","doi":"10.1007/s10772-023-10030-3","title":"Mouth2Audio: intelligible audio synthesis from videos with distinctive vowel articulation","year":2023,"lang":"en","type":"article","venue":"International Journal of Speech Technology","topic":"Speech and Audio Processing","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Simon Fraser University","funders":"Social Sciences and Humanities Research Council of Canada; Simon Fraser University","keywords":"Computer science; Speech recognition; Formant; Spectrogram; Manner of articulation; Landmark; Vowel; Articulation (sociology); Artificial intelligence","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003623043,0.0008194661,0.0004718798,0.0005269389,0.0001836145,0.0006000875,0.0004658747,0.000744897,0.01603915],"category_scores_gemma":[0.0009159148,0.0001815444,0.0002575834,0.0002890641,0.0001973786,0.000411289,0.0007350207,0.0003202604,0.002492446],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001244913,"about_ca_system_score_gemma":0.0001994901,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000776972,"about_ca_topic_score_gemma":0.001874512,"domain_scores_codex":[0.9997992,0.00002995598,0.0000109117,0.00004622735,0.00008658168,0.00002711394],"domain_scores_gemma":[0.9997943,0.00009476554,0.000008249465,0.00002666899,0.00004875722,0.00002730961],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003034633,0.0001252532,0.0008024006,0.0006529271,0.00007065848,0.0007508974,0.0001913971,0.005037567,0.5721171,0.001956093,0.01092339,0.4043376],"study_design_scores_gemma":[0.0008570378,0.001654393,0.007710631,0.0001278205,0.0001169011,0.00223728,0.0002910107,0.2062376,0.7202672,0.002150293,0.05821449,0.0001353723],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.239103,0.001944444,0.7051635,0.0003843819,0.001014264,0.0005244021,0.008740818,0.02363711,0.01948804],"genre_scores_gemma":[0.5264291,0.0008605495,0.4280856,0.0002526491,0.0003611635,0.0004702439,0.01274555,0.002854085,0.02794107],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01603915,"threshold_uncertainty_score":0.05365622,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009332505270161847,"score_gpt":0.2516490338928938,"score_spread":0.242316528622732,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}