{"id":"W4401281176","doi":"10.1016/j.csl.2024.101695","title":"Speech self-supervised representations benchmarking: A case for larger probing heads","year":2024,"lang":"en","type":"article","venue":"Computer Speech & Language","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"ca_institutions":"Concordia University; Mila - Quebec Artificial Intelligence Institute","funders":"Agence de l'innovation de Défense","keywords":"Benchmarking; Computer science; Ranking (information retrieval); Inference; Task (project management); Downstream (manufacturing); Generalization; Feature (linguistics); Set (abstract data type); Artificial intelligence; Architecture; Machine learning; Benchmark (surveying); Data set; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01443455,0.001057798,0.001857942,0.0008149768,0.001273586,0.002517381,0.00396877,0.003240299,0.01096563],"category_scores_gemma":[0.06212056,0.0004530371,0.0006864246,0.001209145,0.001820771,0.005055192,0.004740322,0.002560414,0.002778806],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001060983,"about_ca_system_score_gemma":0.001779757,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004018226,"about_ca_topic_score_gemma":0.006908358,"domain_scores_codex":[0.9876505,0.006068951,0.0006856744,0.002326834,0.002479623,0.000788386],"domain_scores_gemma":[0.9627472,0.01789235,0.0007019229,0.01205339,0.005691091,0.0009140311],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002778876,0.001028681,0.01237425,0.0007342944,0.0003583696,0.0007450592,0.001429851,0.1156187,0.04631171,0.01639373,0.03014494,0.7720816],"study_design_scores_gemma":[0.0002108743,0.00165362,0.01188271,0.000217963,0.0001807859,0.001240989,0.001964666,0.8060158,0.08726761,0.04656624,0.04261881,0.0001798329],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2802281,0.001737212,0.6750554,0.003027526,0.0008945772,0.0005666703,0.0023795,0.0205867,0.01552432],"genre_scores_gemma":[0.8184054,0.0001600215,0.1693239,0.000944551,0.0001398788,0.0003267233,0.004011733,0.002252538,0.004435222],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01443455,"threshold_uncertainty_score":0.07633811,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02313783000429004,"score_gpt":0.2944255563429296,"score_spread":0.2712877263386396,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}