{"id":"W4393619018","doi":"10.14569/ijacsa.2024.0150359","title":"Speech Emotion Recognition in Multimodal Environments with Transformer: Arabic and English Audio Datasets","year":2024,"lang":"en","type":"article","venue":"International Journal of Advanced Computer Science and Applications","topic":"Speech and Audio Processing","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Arabic; Transformer; Speech recognition; Computer science; Natural language processing; Linguistics; Engineering; Electrical engineering; Voltage","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001034271,0.001725982,0.0005419137,0.001374181,0.0005420266,0.0007234007,0.001069617,0.001051837,0.003344566],"category_scores_gemma":[0.002798283,0.0001759122,0.0007159564,0.0007627797,0.0004323027,0.0008779107,0.001485375,0.001188781,0.003525391],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006255377,"about_ca_system_score_gemma":0.0006153273,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01357431,"about_ca_topic_score_gemma":0.0166052,"domain_scores_codex":[0.9990997,0.0002109807,0.00009296535,0.0002129494,0.0002836623,0.00009961577],"domain_scores_gemma":[0.9990911,0.0002059289,0.00005807374,0.0002047145,0.0003378022,0.0001023666],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.005720865,0.003419168,0.02492831,0.001732459,0.0006082972,0.002317862,0.001059751,0.0347919,0.07691372,0.001913866,0.2037851,0.6428087],"study_design_scores_gemma":[0.001175535,0.002898829,0.1626938,0.0003427103,0.0005102343,0.003231772,0.003901065,0.5822374,0.1181941,0.003282241,0.1210456,0.0004866875],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8411518,0.003810923,0.03108828,0.001867765,0.001282378,0.000995098,0.08876044,0.0156988,0.01534462],"genre_scores_gemma":[0.6569484,0.0009939144,0.0490088,0.0005346509,0.0003007377,0.0008893266,0.2808495,0.0003519426,0.01012279],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01357431,"threshold_uncertainty_score":0.02699059,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007168853474817364,"score_gpt":0.2539454191585412,"score_spread":0.2467765656837238,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}