{"id":"W3110458199","doi":"10.48550/arxiv.2011.11588","title":"The Zero Resource Speech Benchmark 2021: Metrics and baselines for\\n unsupervised spoken language modeling","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Benchmark (surveying); Computer science; Zero (linguistics); Spoken language; Speech recognition; Language model; Resource (disambiguation); Natural language processing; Artificial intelligence; Linguistics; Computer network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008466177,0.005035337,0.001801226,0.00429544,0.001733624,0.003369447,0.004034577,0.004251027,0.008236106],"category_scores_gemma":[0.02916052,0.0007093634,0.001345929,0.002553582,0.001610119,0.003882237,0.005221317,0.003079782,0.007929441],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002444305,"about_ca_system_score_gemma":0.002797822,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02022963,"about_ca_topic_score_gemma":0.02302516,"domain_scores_codex":[0.987906,0.004983707,0.001189133,0.002071726,0.003031641,0.0008178097],"domain_scores_gemma":[0.9894451,0.003893943,0.0005395199,0.002935818,0.002624623,0.000560901],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003975391,0.002648874,0.009375365,0.003830004,0.001086266,0.0007030098,0.0006885391,0.1109166,0.02793944,0.01294613,0.2818732,0.5440172],"study_design_scores_gemma":[0.001012396,0.003834008,0.02144711,0.0009142276,0.000477259,0.001553033,0.001316751,0.6758769,0.1142541,0.03458233,0.1441367,0.0005950606],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2563475,0.01523085,0.4457512,0.002925552,0.003632071,0.004064725,0.1185926,0.09531092,0.05814464],"genre_scores_gemma":[0.3260782,0.001741505,0.2848444,0.001128659,0.0004042697,0.005405781,0.3527777,0.00759493,0.02002462],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02022963,"threshold_uncertainty_score":0.04477394,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09401508568117196,"score_gpt":0.2031982157580533,"score_spread":0.1091831300768813,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}