{"id":"W4399434529","doi":"10.1038/s41598-024-63776-4","title":"An enhanced speech emotion recognition using vision transformer","year":2024,"lang":"en","type":"article","venue":"Scientific Reports","topic":"Emotion and Mood Recognition","field":"Psychology","cited_by":63,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Generalizability theory; Computer science; Emotion recognition; Leverage (statistics); Speech recognition; Transformer; Spectrogram; Feature extraction; Artificial intelligence; Benchmark (surveying); Machine learning; Psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004476268,0.0005029922,0.0005396396,0.0004066675,0.000140025,0.0006369156,0.00070332,0.0004764153,0.002521539],"category_scores_gemma":[0.0009347898,0.0001668156,0.0006167023,0.0002516125,0.0002175183,0.0008793682,0.0007936898,0.0007255359,0.001348734],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002993434,"about_ca_system_score_gemma":0.0003182458,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00146659,"about_ca_topic_score_gemma":0.001476784,"domain_scores_codex":[0.9997346,0.00004906467,0.0000113483,0.00008379758,0.0000835526,0.00003765361],"domain_scores_gemma":[0.9998222,0.00004980055,0.000009389014,0.00002529675,0.00007375058,0.00001948911],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006857811,0.0003141391,0.001607335,0.0001509949,0.0001155446,0.0001841101,0.0001035626,0.02306508,0.3142284,0.003162734,0.005911092,0.6504712],"study_design_scores_gemma":[0.00003437759,0.000335712,0.002035453,0.00001282711,0.00006703298,0.0003481956,0.00004113152,0.8974199,0.09506413,0.001624233,0.0029863,0.00003074624],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07520355,0.0006182517,0.9164372,0.000291207,0.0002752681,0.0001178469,0.0002507542,0.003215115,0.003590844],"genre_scores_gemma":[0.7751443,0.0005975566,0.215797,0.0004927067,0.00009571524,0.0001010251,0.0008164391,0.0001510449,0.00680424],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002521539,"threshold_uncertainty_score":0.008435428,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04669083380122481,"score_gpt":0.3621511617437113,"score_spread":0.3154603279424865,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}