{"id":"W4412672023","doi":"10.1101/2025.07.25.25332211","title":"Performance Analysis of Speech Recognition Models in Automated Scoring of the QuickSIN Test","year":2025,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"London Health Sciences Centre; Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Test (biology); Speech recognition; Computer science; Natural language processing; Artificial intelligence; Pattern recognition (psychology); Biology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008740417,0.001053336,0.0006652251,0.001168634,0.0002853214,0.001627598,0.0008454266,0.0006055618,0.002749094],"category_scores_gemma":[0.03656092,0.0003204379,0.0006131722,0.0005504825,0.0003497466,0.001007947,0.001045905,0.0003528816,0.002091746],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005199859,"about_ca_system_score_gemma":0.0004516388,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002697411,"about_ca_topic_score_gemma":0.002572565,"domain_scores_codex":[0.9919506,0.004226014,0.0006490605,0.001314135,0.001643835,0.000216391],"domain_scores_gemma":[0.9675565,0.02261754,0.002089656,0.001837389,0.005315303,0.0005837035],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01454019,0.0008015861,0.314892,0.00059403,0.0006858226,0.0002741335,0.002398827,0.02468931,0.06384716,0.0009373136,0.003649652,0.5726901],"study_design_scores_gemma":[0.0003436092,0.006764602,0.3836206,0.0001672208,0.0005211799,0.001271431,0.001360912,0.5051622,0.09550489,0.000799718,0.004189847,0.000293833],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9342929,0.0005164146,0.05905456,0.00008313813,0.000107331,0.0003206975,0.0005726028,0.001512672,0.003539808],"genre_scores_gemma":[0.9731041,0.0001053721,0.02495462,0.00003863379,0.00002184285,0.0001926496,0.0004893207,0.0001442048,0.0009493005],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008740417,"threshold_uncertainty_score":0.04622424,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04686078083060766,"score_gpt":0.2710023107536934,"score_spread":0.2241415299230858,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}