{"id":"W4392100823","doi":"10.1121/10.0024876","title":"Evaluating OpenAI's Whisper ASR: Performance analysis across diverse accents and speaker traits","year":2024,"lang":"en","type":"article","venue":"JASA Express Letters","topic":"Phonetics and Phonology Research","field":"Psychology","cited_by":66,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Leverhulme Trust","keywords":"Typology; Linguistics; First language; Computer science; American English; Speech recognition; Natural language processing; Psychology; History","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001869041,0.0004740172,0.0005260643,0.0005000007,0.0003141605,0.0009614066,0.0002892872,0.0003687046,0.00190093],"category_scores_gemma":[0.004334999,0.0001256198,0.0002488934,0.0002227829,0.0002710313,0.0004919497,0.0006178719,0.0002210338,0.001775577],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001533669,"about_ca_system_score_gemma":0.0001963348,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0019479,"about_ca_topic_score_gemma":0.003161914,"domain_scores_codex":[0.9991475,0.0002314767,0.00006728653,0.0002182032,0.0002685812,0.00006683386],"domain_scores_gemma":[0.99608,0.00191248,0.0002588533,0.0002865441,0.001117728,0.0003444345],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.004181405,0.0004373577,0.2415194,0.0004311108,0.0003281504,0.0005795166,0.006313195,0.001952715,0.455216,0.0002530823,0.0007945036,0.2879936],"study_design_scores_gemma":[0.00006322475,0.005836932,0.8629349,0.00003685578,0.0002926967,0.002538159,0.004149502,0.01374588,0.1075172,0.0002082111,0.002532917,0.0001435493],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9945465,0.000149722,0.003185885,0.00001814374,0.0000132835,0.00003072249,0.0001545967,0.0001875175,0.001713511],"genre_scores_gemma":[0.9925248,0.000157639,0.004304938,0.0000394085,0.00001473295,0.00002732904,0.0005266189,0.00005723671,0.002347325],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0019479,"threshold_uncertainty_score":0.009884536,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09515323397452029,"score_gpt":0.4407263609346739,"score_spread":0.3455731269601536,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}