{"id":"W4312899222","doi":"10.1109/mlsp55214.2022.9943501","title":"An Efficient Transformer-Based Model for Voice Activity Detection","year":2022,"lang":"en","type":"article","venue":"","topic":"Speech and Audio Processing","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Speech recognition; Transformer; Voice activity detection; Feature extraction; Pattern recognition (psychology); Artificial intelligence; Speech processing; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002392894,0.00006748843,0.00006298793,0.00006476739,0.0004834199,0.00007793618,0.0003328501,0.00001543474,0.000008049737],"category_scores_gemma":[0.000004856251,0.00006438404,0.00004994248,0.0002288778,0.000008322655,0.0002365591,0.00001289846,0.00007786517,0.000001680364],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006680429,"about_ca_system_score_gemma":0.000104627,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001163798,"about_ca_topic_score_gemma":0.00004021371,"domain_scores_codex":[0.999263,0.00002225797,0.00006991354,0.000261424,0.000193823,0.0001895258],"domain_scores_gemma":[0.9996607,0.0000287756,0.00002991749,0.0001952541,0.00003001859,0.00005539149],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002853938,0.000151881,0.000003733162,0.000008979629,0.000001646387,3.47537e-7,0.0002070748,0.6256058,0.1624117,0.00009357319,0.0000134011,0.2114733],"study_design_scores_gemma":[0.0001968942,0.0001108001,0.00001710371,5.029603e-7,0.000001529388,0.000001224028,0.00001302761,0.604435,0.3948831,0.0001734249,0.0001071985,0.0000602646],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1821883,0.000004818184,0.8167881,0.0003288418,0.00009606317,0.0001414134,0.000003008569,0.0001768148,0.0002726746],"genre_scores_gemma":[0.9306893,5.771307e-8,0.06876445,0.0003648293,0.00001391758,0.00008382958,9.188983e-7,0.000005247007,0.00007742119],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7485011,"threshold_uncertainty_score":0.3718124,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02264492892248568,"score_gpt":0.2676246582492399,"score_spread":0.2449797293267542,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}