{"id":"W4391683008","doi":"10.3991/ijim.v18i03.43013","title":"Convolutional Neural Network Architectures for Gender, Emotional Detection from Speech and Speaker Diarization","year":2024,"lang":"en","type":"article","venue":"International Journal of Interactive Mobile Technologies (iJIM)","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Speaker diarisation; Convolutional neural network; Speech recognition; Computer science; Speaker recognition; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004909517,0.0006931244,0.0002609559,0.0003930233,0.0002488799,0.0004377275,0.0007886697,0.000603956,0.002233671],"category_scores_gemma":[0.0009649397,0.0002228806,0.0003496722,0.0003270688,0.0002177156,0.0005146602,0.0005331959,0.000811221,0.000803874],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007002074,"about_ca_system_score_gemma":0.0006035141,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01029124,"about_ca_topic_score_gemma":0.01534282,"domain_scores_codex":[0.9998235,0.00002526262,0.000008555413,0.00005106768,0.00004634204,0.0000452171],"domain_scores_gemma":[0.9998047,0.00005843995,0.00001836536,0.00002424077,0.00008061191,0.00001368966],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0004089627,0.0002368997,0.004504263,0.0001350689,0.0001802382,0.0001672951,0.0001492626,0.2314563,0.06797596,0.008539221,0.005485873,0.6807607],"study_design_scores_gemma":[0.000004733662,0.00005481498,0.001887677,0.00001142958,0.00003102571,0.00004492724,0.00001667792,0.9842235,0.0101592,0.001675414,0.001878318,0.00001223398],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1360346,0.001975565,0.8470092,0.0004513095,0.0002714065,0.00009853139,0.0005326551,0.003220269,0.01040644],"genre_scores_gemma":[0.8435296,0.0009104559,0.1386165,0.0002102389,0.0000649647,0.0001280352,0.001115974,0.00008525015,0.01533888],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01029124,"threshold_uncertainty_score":0.02046269,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01866543782244487,"score_gpt":0.2789606270943671,"score_spread":0.2602951892719222,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}