{"id":"W4410431605","doi":"10.1016/j.csl.2025.101815","title":"BERSting at the screams: A benchmark for distanced, emotional and shouted speech recognition","year":2025,"lang":"en","type":"article","venue":"Computer Speech & Language","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs","keywords":"Computer science; Benchmark (surveying); Speech recognition; Emotion recognition; Natural language processing","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003489852,0.005151149,0.002551382,0.00336269,0.001795651,0.002800253,0.004288053,0.004305886,0.006646022],"category_scores_gemma":[0.008632181,0.0005060268,0.001759357,0.002427632,0.001288913,0.003089201,0.004704046,0.002739167,0.01679813],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001603639,"about_ca_system_score_gemma":0.001604863,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0186673,"about_ca_topic_score_gemma":0.02980642,"domain_scores_codex":[0.9934953,0.001437193,0.0008456407,0.001481279,0.002141322,0.0005992823],"domain_scores_gemma":[0.9950969,0.001296526,0.0003247551,0.001179124,0.001569675,0.0005329923],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003447199,0.002146336,0.008837781,0.005208268,0.0007347199,0.001894395,0.0009301549,0.01627897,0.03183508,0.001799467,0.5668887,0.3599989],"study_design_scores_gemma":[0.00148995,0.004714636,0.1042015,0.001961114,0.0007180106,0.009171566,0.006402718,0.2185847,0.08512628,0.006732907,0.5599101,0.0009865361],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"methods","genre_scores_codex":[0.3626134,0.026698,0.04577526,0.003958049,0.007957797,0.002952316,0.4459036,0.04962753,0.05451411],"genre_scores_gemma":[0.113848,0.00190415,0.03866171,0.0009613886,0.0005086674,0.001184652,0.8280141,0.001057448,0.01385982],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.0186673,"threshold_uncertainty_score":0.0371173,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01565752871258395,"score_gpt":0.2604487816432463,"score_spread":0.2447912529306624,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}