{"id":"W4413018410","doi":"10.1109/iv64158.2025.11097450","title":"Conditional Transformer-Based U-Net Architecture for Speech Emotion Recognition","year":2025,"lang":"en","type":"article","venue":"","topic":"Speech and Audio Processing","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Intertek (Canada)","funders":"","keywords":"Computer science; Speech recognition; Transformer; Architecture; Natural language processing; Artificial intelligence; Engineering; Electrical engineering; Voltage","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001322992,0.00008663312,0.00008365473,0.0001445511,0.0001429567,0.0001140765,0.0002030355,0.00005428019,0.00005748773],"category_scores_gemma":[0.00002928972,0.000076539,0.00006978564,0.0002792524,0.00002242512,0.0002156295,0.000007074195,0.00007093189,0.00001722221],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002468919,"about_ca_system_score_gemma":0.0001330895,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000003041799,"about_ca_topic_score_gemma":0.00001284751,"domain_scores_codex":[0.9993177,0.00001460151,0.0001352001,0.0002387904,0.0001205948,0.0001731122],"domain_scores_gemma":[0.9996325,0.00009087942,0.00002990018,0.0001142451,0.0000976064,0.00003483822],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00002111179,0.00006052528,0.00003789348,0.00007918788,0.00001148722,0.000001373158,0.00003745116,0.0001471891,0.009839826,0.003020608,0.00540709,0.9813362],"study_design_scores_gemma":[0.0008820423,0.00007580104,0.0003761825,0.00006162957,0.000008668065,0.000006377643,0.00001131885,0.008152993,0.7718617,0.2084526,0.009974,0.0001367376],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005582812,0.00002025734,0.9773083,0.008370815,0.0001752984,0.0002236759,0.00001769434,0.0001778999,0.008123265],"genre_scores_gemma":[0.297444,0.000001873939,0.6946911,0.00663018,0.00009056545,0.00005741879,0.0002053335,0.000006262214,0.0008732633],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9811995,"threshold_uncertainty_score":0.312117,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01662735401318722,"score_gpt":0.2626648908498537,"score_spread":0.2460375368366665,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}