{"id":"W4399748891","doi":"10.20944/preprints202404.1456.v1","title":"Combining Transformer, CNN, and LSTM Architectures: A Novel Ensemble Learning Technique That Leverages Multi-acoustic Features for Speech Emotion Recognition in Distance Education Classrooms","year":2024,"lang":"en","type":"preprint","venue":"Preprints.org","topic":"Emotion and Mood Recognition","field":"Psychology","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Speech recognition; Spectrogram; Transformer; Mel-frequency cepstrum; Deep learning; Artificial intelligence; Convolutional neural network; Feature extraction; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006844466,0.001019954,0.0005278962,0.0005243032,0.0002467555,0.0004750725,0.0008176733,0.0005235518,0.001039968],"category_scores_gemma":[0.001035319,0.0003037931,0.0007236442,0.0004114957,0.0001973933,0.001132471,0.0008307691,0.001111888,0.0006168969],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000364218,"about_ca_system_score_gemma":0.0004982473,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004749102,"about_ca_topic_score_gemma":0.007258743,"domain_scores_codex":[0.9997222,0.00004408228,0.00001431546,0.00008614419,0.00008420165,0.00004901907],"domain_scores_gemma":[0.9997266,0.00006719064,0.00002092623,0.00003457467,0.000130984,0.00001966152],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002113882,0.0001902383,0.003189839,0.00006756748,0.0002103345,0.0001528905,0.00009978135,0.1510926,0.04630796,0.001517216,0.003342558,0.7936176],"study_design_scores_gemma":[0.00000299402,0.00007618971,0.0005529169,0.000005495283,0.00003968337,0.00004099111,0.00001666233,0.9899321,0.007891334,0.0006412783,0.0007926039,0.000007882987],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.095071,0.0008696555,0.8971464,0.0002471409,0.000189293,0.00005598014,0.0001320309,0.002644538,0.003643929],"genre_scores_gemma":[0.8453184,0.0005673491,0.1470119,0.0002440296,0.0000766201,0.00006441768,0.0004924455,0.0001392059,0.006085691],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004749102,"threshold_uncertainty_score":0.009442866,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1372193832998145,"score_gpt":0.3841146379727956,"score_spread":0.2468952546729811,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}