{"id":"W4389518357","doi":"10.18653/v1/2023.arabicnlp-1.38","title":"VoxArabica: A Robust Dialect-Aware Arabic Speech Recognition System","year":2023,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Arabic; Computer science; Upload; Natural language processing; Interface (matter); Speech recognition; Range (aeronautics); Artificial intelligence; Language model; Linguistics; World Wide Web; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.000493311,0.0001791481,0.0002228937,0.000370218,0.0001902609,0.000242985,0.0005640298,0.0001136717,0.0003580628],"category_scores_gemma":[0.0001068341,0.0001585758,0.0001355502,0.001507683,0.0000295807,0.0005128013,0.0001434728,0.0001318004,0.01491347],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008569067,"about_ca_system_score_gemma":0.00005783115,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004071101,"about_ca_topic_score_gemma":0.00003117434,"domain_scores_codex":[0.9982337,0.0001334728,0.0003029802,0.0005313905,0.0003854015,0.0004130548],"domain_scores_gemma":[0.9988699,0.0002354851,0.00007724969,0.0004909475,0.0001648609,0.0001615605],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000007342566,0.00005733054,0.0001581633,0.00007658731,0.00003651806,0.0002551021,0.0001343527,0.00001324436,0.0004853751,0.002186697,0.02727837,0.9693109],"study_design_scores_gemma":[0.003269997,0.0004804062,0.01573339,0.001017113,0.0001247967,0.001706963,0.002970736,0.7036155,0.228868,0.01212014,0.0264779,0.003615065],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1636922,0.0000511518,0.5554057,0.005589459,0.003466448,0.001159382,0.00004771012,0.01402445,0.2565636],"genre_scores_gemma":[0.9025097,0.00007652818,0.08582593,0.001080317,0.0004944174,0.0002320297,0.00007814069,0.0000578369,0.009645034],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9656959,"threshold_uncertainty_score":0.9858536,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05897081501115145,"score_gpt":0.234406232164648,"score_spread":0.1754354171534966,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}