{"id":"W4410922640","doi":"10.1177/23312165251347131","title":"Language-agnostic, Automated Assessment of Listeners’ Speech Recall Using Large Language Models","year":2025,"lang":"en","type":"article","venue":"Trends in Hearing","topic":"Interpreting and Communication in Healthcare","field":"Health Professions","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Baycrest Hospital; University of Toronto","funders":"University of Toronto; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Recall; Computer science; Comprehension; Natural language processing; Context (archaeology); Spoken language; Similarity (geometry); Conversation; Speech perception; Linguistics; Psychology; Speech recognition; Artificial intelligence; Cognitive psychology; Perception; Communication","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001482985,0.000887785,0.0003527694,0.0006055136,0.0001654782,0.001142979,0.0004567515,0.0004391793,0.002180551],"category_scores_gemma":[0.007312502,0.000195819,0.000594953,0.000263702,0.0002517574,0.00120102,0.0009999875,0.000528098,0.001531764],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000220393,"about_ca_system_score_gemma":0.0002811246,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007942962,"about_ca_topic_score_gemma":0.001688533,"domain_scores_codex":[0.9992117,0.0003194691,0.00006448884,0.0002117904,0.0001606592,0.00003189228],"domain_scores_gemma":[0.9976271,0.001227341,0.000259517,0.0003483234,0.0004567031,0.00008104061],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001560766,0.0005489436,0.05979352,0.0006429809,0.0004853178,0.0004145629,0.003126217,0.03483081,0.2168431,0.001516424,0.003234803,0.6770026],"study_design_scores_gemma":[0.0001086261,0.002228874,0.1506409,0.0001391514,0.0003830211,0.001825287,0.002496225,0.6689675,0.1553863,0.009001271,0.008517817,0.0003049342],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6527024,0.0002935991,0.3407676,0.0001176084,0.00006602173,0.0001585237,0.001124526,0.002646714,0.00212302],"genre_scores_gemma":[0.9226366,0.0001519862,0.07379062,0.00004594353,0.00002767202,0.000187689,0.001675613,0.0002095333,0.001274356],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.002180551,"threshold_uncertainty_score":0.007842839,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1178072682074484,"score_gpt":0.5206685656166137,"score_spread":0.4028612974091653,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}