{"id":"W4405878042","doi":"10.54097/6pzrwp08","title":"Multimodal Speech Emotion Recognition via Transformer-Based Hybrid Fusion and Dual Cross-entropy Techniques","year":2024,"lang":"en","type":"article","venue":"Highlights in Science Engineering and Technology","topic":"Emotion and Mood Recognition","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Cross entropy; Transformer; Speech recognition; Emotion recognition; Architecture; Hyperparameter; Artificial intelligence; Modalities; Multimodality; Scalability; Computation; Machine learning; Pattern recognition (psychology); Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004252958,0.0001668753,0.0001517187,0.001510318,0.0001323843,0.00009349379,0.00009392233,0.000208747,0.00004739668],"category_scores_gemma":[0.00003938715,0.0001493817,0.00002383754,0.0009318703,0.0004395396,0.0002627676,0.00002239064,0.0002957872,0.00007565667],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006066499,"about_ca_system_score_gemma":0.00002854368,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000277465,"about_ca_topic_score_gemma":0.000006557073,"domain_scores_codex":[0.9987007,0.00001577627,0.0002412083,0.0005401822,0.0001544279,0.000347715],"domain_scores_gemma":[0.9996539,0.0000444194,0.0000265688,0.0001423358,0.00006412933,0.00006862323],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0000300497,0.0001295486,0.0003612898,0.0001060463,0.00001134755,0.0001956493,0.0003308196,0.00001580195,0.3202947,0.02017938,0.0000549687,0.6582904],"study_design_scores_gemma":[0.001203991,0.0006175698,0.005873669,0.0005404254,0.00002905885,0.001046983,0.0001401321,0.03941235,0.9354155,0.007360756,0.007728834,0.0006307104],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9752349,0.0002970572,0.02042986,0.001279364,0.000953383,0.0002711838,0.00001514279,0.0009486268,0.0005704894],"genre_scores_gemma":[0.9933167,0.0001553946,0.006291687,0.00002192131,0.00007299522,0.00005101815,0.00001512604,0.00001760517,0.00005753082],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6576596,"threshold_uncertainty_score":0.6091607,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01183205598505762,"score_gpt":0.2783163061964723,"score_spread":0.2664842502114147,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}