{"id":"W4413018410","doi":"10.1109/iv64158.2025.11097450","title":"Conditional Transformer-Based U-Net Architecture for Speech Emotion Recognition","year":2025,"lang":"en","type":"article","venue":"","topic":"Speech and Audio Processing","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Intertek (Canada)","funders":"","keywords":"Computer science; Speech recognition; Transformer; Architecture; Natural language processing; Artificial intelligence; Engineering; Electrical engineering; Voltage","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003972567,0.0007063763,0.0004087965,0.0003597406,0.0002295207,0.0005373373,0.001211902,0.0005232719,0.005834736],"category_scores_gemma":[0.0006634526,0.0002090327,0.0005420893,0.0002505305,0.0003622527,0.001023655,0.0007718127,0.0009350873,0.002097655],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006013543,"about_ca_system_score_gemma":0.0006158536,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004532804,"about_ca_topic_score_gemma":0.007056187,"domain_scores_codex":[0.999795,0.00003232877,0.00001233983,0.00006641352,0.00004802346,0.00004584821],"domain_scores_gemma":[0.9998307,0.00004948241,0.00001261875,0.00002357471,0.00006575233,0.00001779082],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001203486,0.0003302558,0.002214606,0.0002168357,0.0001540732,0.0003071036,0.000130738,0.09552771,0.1003832,0.02038382,0.01323853,0.7659097],"study_design_scores_gemma":[0.00001058519,0.0001255498,0.0004626529,0.00001123658,0.00003498949,0.00009550743,0.00001780344,0.9673588,0.02488835,0.004735086,0.002244477,0.00001489162],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03011398,0.0007374084,0.9576284,0.0001852451,0.0002059159,0.0000788051,0.0002763162,0.005138356,0.005635743],"genre_scores_gemma":[0.7550048,0.0005166408,0.2312333,0.0004797939,0.00009221223,0.0001653673,0.001258532,0.0002064116,0.01104293],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005834736,"threshold_uncertainty_score":0.01951909,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01662735401318722,"score_gpt":0.2626648908498537,"score_spread":0.2460375368366665,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}