{"id":"W4415571224","doi":"10.7717/peerj-cs.3292","title":"Hybrid-Module Transformer: enhancing speech emotion recognition with HuBERT, LSTM, and ResNet-50","year":2025,"lang":"en","type":"article","venue":"PeerJ Computer Science","topic":"Emotion and Mood Recognition","field":"Psychology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Key Research and Development Program of China; Natural Science Foundation of Fujian Province; National Natural Science Foundation of China","keywords":"Spectrogram; Generalizability theory; Emotion recognition; Mel-frequency cepstrum; Artificial neural network; Feature (linguistics); Transformer; Cepstrum; Benchmark (surveying)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005714507,0.001262718,0.0005468255,0.0004354848,0.0001860229,0.0005393609,0.001126259,0.0006100077,0.002271428],"category_scores_gemma":[0.0008874045,0.0002809303,0.000769741,0.0002506906,0.0003121682,0.001391836,0.0007725621,0.0009957881,0.00165972],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005174135,"about_ca_system_score_gemma":0.0005029564,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00328984,"about_ca_topic_score_gemma":0.004877294,"domain_scores_codex":[0.9997978,0.00003815903,0.000009647782,0.00007564493,0.00004371168,0.00003507595],"domain_scores_gemma":[0.9998496,0.00003973168,0.00001235756,0.00002339678,0.00005817016,0.00001669912],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006041568,0.0005598848,0.002237111,0.0002084931,0.0002904134,0.0003714687,0.0002602894,0.1814655,0.1080779,0.005574436,0.0154893,0.684861],"study_design_scores_gemma":[0.00001369105,0.00008863761,0.000436888,0.000005245491,0.00004443364,0.00004697368,0.00001624126,0.9801257,0.01629159,0.001554494,0.001363897,0.00001213196],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1151493,0.0009887562,0.8606548,0.00044197,0.0003169074,0.000213069,0.0004263079,0.0149949,0.006813889],"genre_scores_gemma":[0.816866,0.0005403797,0.1654581,0.0005022029,0.0001261048,0.000205727,0.001339366,0.0005443127,0.01441778],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00328984,"threshold_uncertainty_score":0.007598639,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01787606752980124,"score_gpt":0.2851272987289677,"score_spread":0.2672512311991665,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}