{"id":"W4404783121","doi":"10.18653/v1/2024.emnlp-main.30","title":"EmphAssess : a Prosodic Benchmark on Assessing Emphasis Transfer in Speech-to-Speech Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Speech and dialogue systems","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Agence Nationale de la Recherche; École des Hautes Etudes en Sciences Sociales; Canadian Institute for Advanced Research","keywords":"Emphasis (telecommunications); Computer science; Benchmark (surveying); Speech recognition; Natural language processing; Artificial intelligence; Telecommunications","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007709055,0.003484631,0.000944055,0.002136757,0.0008721083,0.002513815,0.00218472,0.002414271,0.005933229],"category_scores_gemma":[0.02820204,0.000510635,0.0009479391,0.0008780994,0.0008376401,0.003011106,0.002991704,0.002139203,0.002552868],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008696018,"about_ca_system_score_gemma":0.001220287,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004881146,"about_ca_topic_score_gemma":0.005966309,"domain_scores_codex":[0.9954608,0.002014703,0.0003977501,0.0007310641,0.001179418,0.0002162363],"domain_scores_gemma":[0.9883261,0.007177868,0.0006968582,0.001172612,0.002041466,0.0005850014],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004052553,0.001365842,0.01891255,0.002711857,0.001139155,0.0007439962,0.00121646,0.4034856,0.08492686,0.006404527,0.04755512,0.4274855],"study_design_scores_gemma":[0.0004737845,0.003482521,0.01744228,0.0003134723,0.0003215329,0.0007431419,0.0005823899,0.889735,0.05591394,0.01004622,0.02071356,0.0002321843],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3692548,0.007023367,0.5331693,0.00127509,0.00182281,0.001564219,0.02062529,0.02769852,0.03756664],"genre_scores_gemma":[0.729133,0.001552689,0.2068358,0.0008521301,0.0004250822,0.001606082,0.04736651,0.003645906,0.00858287],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007709055,"threshold_uncertainty_score":0.04076988,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03559975349177633,"score_gpt":0.2959456531634647,"score_spread":0.2603458996716883,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}