{"id":"W2404260248","doi":"10.21437/interspeech.2013-610","title":"Automatic human utility evaluation of ASR systems: does WER really predict performance?","year":2013,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canada Research Chairs; University of Toronto","funders":"","keywords":"Metric (unit); Computer science; Task (project management); Context (archaeology); Audit; Word error rate; Word (group theory); Speech recognition; Performance metric; Artificial intelligence; Natural language processing; Range (aeronautics); Machine learning; Mathematics; Engineering; Operations management","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01741584,0.001668913,0.001156211,0.00212234,0.0005246754,0.002529523,0.00115693,0.001884682,0.00244526],"category_scores_gemma":[0.07993484,0.0003148357,0.0004253942,0.001223402,0.001707448,0.003682839,0.001675426,0.0009497075,0.002888532],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005229698,"about_ca_system_score_gemma":0.0005463801,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001318193,"about_ca_topic_score_gemma":0.00151315,"domain_scores_codex":[0.9740345,0.01735868,0.00147738,0.002019137,0.004509061,0.0006012638],"domain_scores_gemma":[0.9269429,0.05206887,0.003855091,0.006993619,0.009183033,0.0009565115],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00370975,0.0007271668,0.1331857,0.0009992418,0.0008963721,0.0006562578,0.002316794,0.05055661,0.08429689,0.01113165,0.009076894,0.7024467],"study_design_scores_gemma":[0.0003122307,0.006381577,0.2287594,0.0003415785,0.0005050899,0.002857191,0.002765389,0.48347,0.2036377,0.05204017,0.01832423,0.0006054649],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5775083,0.003328635,0.3908765,0.001166189,0.0003782273,0.0002845293,0.001186286,0.003900477,0.02137081],"genre_scores_gemma":[0.9642302,0.0003229638,0.03188057,0.0001892518,0.00009625817,0.0001253386,0.0007322059,0.0003890753,0.002034233],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01741584,"threshold_uncertainty_score":0.09210485,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04054442211315303,"score_gpt":0.273337642310252,"score_spread":0.232793220197099,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}