{"id":"W4391683008","doi":"10.3991/ijim.v18i03.43013","title":"Convolutional Neural Network Architectures for Gender, Emotional Detection from Speech and Speaker Diarization","year":2024,"lang":"en","type":"article","venue":"International Journal of Interactive Mobile Technologies (iJIM)","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Speaker diarisation; Convolutional neural network; Speech recognition; Computer science; Speaker recognition; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002884093,0.0001930884,0.0002157188,0.0005300198,0.00009127933,0.0004033072,0.000741266,0.0001373126,0.00007403918],"category_scores_gemma":[0.0005099219,0.0001623679,0.0002074593,0.0002205622,0.0001314704,0.0005002929,0.0002316668,0.0004217828,0.00001236472],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002207643,"about_ca_system_score_gemma":0.00007455842,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001174813,"about_ca_topic_score_gemma":0.00001288351,"domain_scores_codex":[0.9984162,0.00006378417,0.0004885249,0.0003485256,0.0004852656,0.0001977109],"domain_scores_gemma":[0.9977984,0.001069996,0.0002888663,0.0001447925,0.0006518473,0.00004607392],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002009245,0.00009500082,0.0005543341,0.00001285612,0.0005836026,0.00012219,0.0002625736,0.001750703,0.008062587,0.004371413,0.001095739,0.9828881],"study_design_scores_gemma":[0.001536353,0.0007725952,0.01189291,0.0005904033,0.000102981,0.003172737,0.00103903,0.534498,0.1188869,0.3084304,0.01838787,0.0006898333],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3121472,0.001376444,0.6799152,0.001736768,0.003961867,0.0002573737,0.00007969529,0.0003316934,0.000193817],"genre_scores_gemma":[0.954808,0.0001173642,0.04418309,0.0001412562,0.0006415805,0.00004555794,0.00001429611,0.00001515337,0.00003368083],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9821982,"threshold_uncertainty_score":0.6621172,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01866543782244487,"score_gpt":0.2789606270943671,"score_spread":0.2602951892719222,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}