{"id":"W4385822952","doi":"10.21437/interspeech.2023-1087","title":"Speech Self-Supervised Representation Benchmarking: Are We Doing it Right?","year":2023,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute","funders":"Grand Équipement National De Calcul Intensif","keywords":"Benchmarking; Computer science; Representation (politics); Artificial intelligence; Speech recognition; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01911605,0.001986068,0.002878993,0.001006231,0.001051898,0.005110569,0.003094519,0.004209173,0.01417939],"category_scores_gemma":[0.04511309,0.0005529828,0.001178984,0.000875955,0.001681962,0.007199361,0.003082347,0.004756134,0.008960153],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001074558,"about_ca_system_score_gemma":0.002060797,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002635327,"about_ca_topic_score_gemma":0.004022419,"domain_scores_codex":[0.9893404,0.004605273,0.0004446271,0.002360982,0.00242188,0.0008268832],"domain_scores_gemma":[0.9796084,0.006041099,0.000669081,0.004262053,0.007618417,0.001801025],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001060166,0.0006251681,0.006748667,0.0005896247,0.0004749422,0.00009201556,0.0001843115,0.01974397,0.01433633,0.006478272,0.116134,0.8335325],"study_design_scores_gemma":[0.0003594106,0.002050158,0.01399572,0.001000845,0.0004484669,0.0006072943,0.0009814961,0.728712,0.06225867,0.0779788,0.1113003,0.0003067317],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07974807,0.01635781,0.8001277,0.04064079,0.01126039,0.0004708155,0.004086777,0.02460221,0.02270547],"genre_scores_gemma":[0.6131751,0.005332646,0.3003003,0.01224956,0.006837318,0.0004627545,0.02042206,0.006841626,0.03437851],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.980884,"threshold_uncertainty_score":0.1010965,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05260799574573499,"score_gpt":0.2907351813080752,"score_spread":0.2381271855623402,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}