{"id":"W2168166926","doi":"10.64152/10125/44034","title":"Establishing a methodology for benchmarking speech synthesis for computer-assisted language learning (CALL)","year":2005,"lang":"en","type":"article","venue":"Language learning & technology","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Benchmarking; Computer science; Benchmark (surveying); Pronunciation; Context (archaeology); Speech synthesis; Task (project management); Set (abstract data type); Artificial intelligence; Natural language processing; Programming language; Linguistics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04650227,0.001718598,0.001550101,0.007523623,0.001281829,0.004229482,0.003182494,0.00208205,0.002854964],"category_scores_gemma":[0.09636955,0.0006509601,0.001233353,0.004195832,0.001673027,0.002851532,0.003404838,0.001712046,0.001501691],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003266777,"about_ca_system_score_gemma":0.004291424,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004309594,"about_ca_topic_score_gemma":0.004503138,"domain_scores_codex":[0.9420938,0.03039026,0.006823642,0.003137367,0.01636702,0.001187813],"domain_scores_gemma":[0.9198712,0.03207025,0.007190705,0.009079143,0.03071632,0.001072479],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001023891,0.002770752,0.03282032,0.004252399,0.0005418852,0.0006159278,0.00417581,0.1215667,0.07622021,0.03160559,0.006528911,0.7178776],"study_design_scores_gemma":[0.0005327129,0.01150177,0.06486069,0.002074452,0.0004595488,0.001293334,0.007786089,0.5696086,0.2418169,0.03342807,0.06601185,0.0006260828],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.08592806,0.0007863762,0.8891461,0.0003354863,0.00015063,0.007948305,0.00124331,0.003570139,0.01089159],"genre_scores_gemma":[0.218857,0.0002705251,0.7705984,0.000105515,0.00003105687,0.005814817,0.002237369,0.0003992402,0.001686089],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.04650227,"threshold_uncertainty_score":0.2459305,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03088655784593764,"score_gpt":0.2995842815383971,"score_spread":0.2686977236924594,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}