{"id":"W4389524018","doi":"10.18653/v1/2023.emnlp-main.182","title":"Generative Spoken Language Model based on continuous word-sized audio tokens","year":2023,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Grand Équipement National De Calcul Intensif; Agence Nationale de la Recherche; Canadian Institute for Advanced Research","keywords":"Computer science; Natural language processing; Language model; Word (group theory); Speech recognition; Artificial intelligence; Generative grammar; Spoken language; Generative model; Bridging (networking); Linguistics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0002236847,0.0001321518,0.0001760466,0.0001873309,0.00009281471,0.0001265124,0.0003923238,0.00005360688,0.0004034839],"category_scores_gemma":[0.0001000028,0.0001072701,0.00009604045,0.000467444,0.00002075977,0.0001323615,0.00007504071,0.00008468959,0.001639469],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000292714,"about_ca_system_score_gemma":0.0000671773,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000177037,"about_ca_topic_score_gemma":0.0000243098,"domain_scores_codex":[0.9988919,0.00007138566,0.0001524675,0.000338364,0.000277607,0.0002682935],"domain_scores_gemma":[0.9991736,0.0002122595,0.00003902896,0.0004185156,0.00005604385,0.0001005076],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007932885,0.000296892,0.0001454451,0.00001653867,0.0000708097,0.0005052831,0.003253593,0.01021681,0.01976442,0.02092331,0.1220952,0.8226324],"study_design_scores_gemma":[0.0004455358,0.00002745274,0.0001949293,0.000009858609,0.0000033691,0.000001731934,0.0001035835,0.9650495,0.03288324,0.0005340421,0.0005853027,0.0001614258],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06584118,0.000007768737,0.8122622,0.005074455,0.0002648239,0.0002616141,0.0000170821,0.001440058,0.1148308],"genre_scores_gemma":[0.686569,0.000009382259,0.2704637,0.008762923,0.0001096588,0.00006991562,0.00002055272,0.0000246968,0.03397022],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9548327,"threshold_uncertainty_score":0.9991379,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02701431847517769,"score_gpt":0.2647048678193151,"score_spread":0.2376905493441374,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}