{"id":"W4312899222","doi":"10.1109/mlsp55214.2022.9943501","title":"An Efficient Transformer-Based Model for Voice Activity Detection","year":2022,"lang":"en","type":"article","venue":"","topic":"Speech and Audio Processing","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Speech recognition; Transformer; Voice activity detection; Feature extraction; Pattern recognition (psychology); Artificial intelligence; Speech processing; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002618308,0.0005408411,0.0005553632,0.0003264855,0.0001810229,0.0006032547,0.001263632,0.0006371657,0.002417034],"category_scores_gemma":[0.0007341603,0.0002745413,0.0005690969,0.0003370948,0.0003159983,0.0008464128,0.0005481753,0.0008706113,0.001302698],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004447255,"about_ca_system_score_gemma":0.000558976,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002685751,"about_ca_topic_score_gemma":0.002602459,"domain_scores_codex":[0.9998356,0.0000283827,0.000008605694,0.00004730055,0.00005367215,0.00002650058],"domain_scores_gemma":[0.9998547,0.00005367671,0.00001154448,0.00001980192,0.00004907637,0.00001115954],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004397235,0.0001346742,0.0009762087,0.0001908373,0.0000848544,0.0002218973,0.00007837817,0.5591915,0.08313852,0.02558836,0.003759181,0.3261958],"study_design_scores_gemma":[0.000003457951,0.00002323555,0.00004782074,0.000002685591,0.000006329654,0.00005644159,0.000002259996,0.9948041,0.003036305,0.001328025,0.0006847256,0.000004620675],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004608115,0.0001881716,0.9935695,0.00004835219,0.00003862708,0.00001935522,0.00005864251,0.0004425856,0.001026717],"genre_scores_gemma":[0.7316033,0.0007938687,0.2591853,0.0001886713,0.00008640952,0.0001553267,0.0004226535,0.0001285794,0.007435982],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002685751,"threshold_uncertainty_score":0.008085787,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02264492892248568,"score_gpt":0.2676246582492399,"score_spread":0.2449797293267542,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}