{"id":"W4412672023","doi":"10.1101/2025.07.25.25332211","title":"Performance Analysis of Speech Recognition Models in Automated Scoring of the QuickSIN Test","year":2025,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"London Health Sciences Centre; Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Test (biology); Speech recognition; Computer science; Natural language processing; Artificial intelligence; Pattern recognition (psychology); Biology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008150469,0.0001876558,0.00056331,0.001064846,0.00003862541,0.00003467319,0.001262401,0.000176758,0.00002976598],"category_scores_gemma":[0.0002919679,0.0001529266,0.0002938519,0.002762295,0.0000617098,0.000183241,0.0007651527,0.0003039339,0.000004508797],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006648312,"about_ca_system_score_gemma":0.0001945723,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000213887,"about_ca_topic_score_gemma":0.0001485785,"domain_scores_codex":[0.9980986,0.0001670646,0.0006909963,0.0004538084,0.0004015764,0.0001880265],"domain_scores_gemma":[0.9979609,0.0003280776,0.0004578649,0.0009596716,0.0002608045,0.00003271218],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002005851,0.0004607853,0.407136,0.001066826,0.0007846242,0.00001047512,0.001398053,0.02986863,0.002973806,0.0001674356,0.0000601289,0.5560532],"study_design_scores_gemma":[0.00009359447,0.000008924132,0.1380962,0.0006787574,0.0001660157,8.887141e-7,0.00001318394,0.8099347,0.05013562,0.0007414461,0.000003831428,0.000126769],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9885569,0.00004453324,0.005386591,0.0001680423,0.0003374772,0.0002498524,0.00006445761,0.0001439926,0.005048121],"genre_scores_gemma":[0.9802931,0.0001373688,0.01937546,0.00003711997,0.0000134856,0.00002876746,0.00001135675,0.000005787558,0.00009758578],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7800661,"threshold_uncertainty_score":0.6236166,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04686078083060766,"score_gpt":0.2710023107536934,"score_spread":0.2241415299230858,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}