{"id":"W4405878042","doi":"10.54097/6pzrwp08","title":"Multimodal Speech Emotion Recognition via Transformer-Based Hybrid Fusion and Dual Cross-entropy Techniques","year":2024,"lang":"en","type":"article","venue":"Highlights in Science Engineering and Technology","topic":"Emotion and Mood Recognition","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Cross entropy; Transformer; Speech recognition; Emotion recognition; Architecture; Hyperparameter; Artificial intelligence; Modalities; Multimodality; Scalability; Computation; Machine learning; Pattern recognition (psychology); Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001478113,0.0008773734,0.0007455212,0.0009964957,0.0002687292,0.0008822011,0.0007077887,0.0005336204,0.001742844],"category_scores_gemma":[0.001753349,0.0002476509,0.001319754,0.0005497849,0.0003507974,0.001448071,0.001465042,0.0009321669,0.0009453915],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004221516,"about_ca_system_score_gemma":0.0003251657,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001212841,"about_ca_topic_score_gemma":0.00123565,"domain_scores_codex":[0.9994187,0.0001588687,0.00004012867,0.0001492826,0.0001637285,0.00006929276],"domain_scores_gemma":[0.9995865,0.0001455,0.00003465452,0.00004587342,0.0001627848,0.00002470288],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000692677,0.0003169137,0.003264583,0.0001154125,0.0002865347,0.000142387,0.0001873451,0.1168426,0.06765252,0.004984181,0.002944714,0.8025703],"study_design_scores_gemma":[0.000007680728,0.0001215518,0.001489818,0.000009562787,0.00005305478,0.00008005488,0.00003045224,0.9832788,0.01207678,0.002182447,0.0006529005,0.00001695158],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0459177,0.0006552166,0.950089,0.0001997607,0.00009077593,0.00006669795,0.0001057663,0.0009666702,0.001908414],"genre_scores_gemma":[0.8077574,0.0007039094,0.185151,0.0002343246,0.0001142204,0.0001373046,0.0006302466,0.0001502932,0.005121292],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.001742844,"threshold_uncertainty_score":0.00781709,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01183205598505762,"score_gpt":0.2783163061964723,"score_spread":0.2664842502114147,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}