{"id":"W4392100823","doi":"10.1121/10.0024876","title":"Evaluating OpenAI's Whisper ASR: Performance analysis across diverse accents and speaker traits","year":2024,"lang":"en","type":"article","venue":"JASA Express Letters","topic":"Phonetics and Phonology Research","field":"Psychology","cited_by":66,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Leverhulme Trust","keywords":"Typology; Linguistics; First language; Computer science; American English; Speech recognition; Natural language processing; Psychology; History","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0006471445,0.0002008096,0.0002808192,0.0002150148,0.0002090796,0.0002523261,0.0003689173,0.0001281986,0.002077917],"category_scores_gemma":[0.0000243811,0.0001790934,0.0001241743,0.0005414113,0.0002048753,0.0002250257,0.0002843652,0.000442513,0.0005046909],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003887341,"about_ca_system_score_gemma":0.00001517398,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001744692,"about_ca_topic_score_gemma":0.00002847127,"domain_scores_codex":[0.9979402,0.0001848921,0.0002534324,0.000655797,0.0003259709,0.0006396691],"domain_scores_gemma":[0.9992535,0.0001502052,0.00004430481,0.0003788635,0.00003888233,0.0001343194],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.0006418393,0.0002995964,0.2441256,0.0003093199,0.006664555,0.0007434178,0.1337465,0.001573604,0.3099944,0.0001923949,0.08621628,0.2154925],"study_design_scores_gemma":[0.001095238,0.0001194412,0.9716848,0.00006267044,0.0003482296,0.00002575104,0.0008003073,0.01077081,0.0008940017,0.00002589518,0.01375205,0.0004208184],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9955963,0.0007311897,0.000228263,0.001282993,0.0006144242,0.0002268911,0.00006828835,0.00006571456,0.001185914],"genre_scores_gemma":[0.9963537,0.00005978862,0.0002495519,0.001231373,0.0001948574,0.00006367192,0.00002263487,0.00002937311,0.001795044],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7275592,"threshold_uncertainty_score":0.9988343,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09515323397452029,"score_gpt":0.4407263609346739,"score_spread":0.3455731269601536,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}