{"id":"W4402980134","doi":"10.1109/otcon60325.2024.10687796","title":"Merging Speech Recognition Capabilities with Large Language Models for Enhanced Communication and Interaction","year":2024,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Horizon College and Seminary","funders":"","keywords":"Computer science; Speech recognition; Natural language processing; Language model; Human–computer interaction; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002050539,0.001078249,0.001051251,0.0006337772,0.0003552809,0.002340904,0.0009766193,0.0008485153,0.007151038],"category_scores_gemma":[0.004631182,0.0004451376,0.001309858,0.0004145259,0.0007174829,0.003681708,0.001921146,0.00145599,0.003030155],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007426363,"about_ca_system_score_gemma":0.0009245939,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002486456,"about_ca_topic_score_gemma":0.003323838,"domain_scores_codex":[0.9987603,0.000424646,0.0001021792,0.0002646619,0.0003619625,0.00008630272],"domain_scores_gemma":[0.9964029,0.001906515,0.000127996,0.0007656314,0.0007048559,0.00009213602],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0006674805,0.0002445342,0.002467906,0.0006009029,0.000313915,0.0004347858,0.000696813,0.2706015,0.07970942,0.04278242,0.008339023,0.5931413],"study_design_scores_gemma":[0.00002295604,0.0002024155,0.0004794405,0.00003941139,0.00009931342,0.0002135352,0.0001203946,0.9372163,0.02600941,0.02190924,0.01363797,0.00004958981],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02080621,0.00055968,0.9670696,0.0006268944,0.0001706667,0.00006813312,0.0001860182,0.005272613,0.00524016],"genre_scores_gemma":[0.5476178,0.001051621,0.4375603,0.0004903236,0.0002208141,0.0001995146,0.001027175,0.0008930121,0.01093943],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007151038,"threshold_uncertainty_score":0.02392262,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02777832604551551,"score_gpt":0.2801247445404287,"score_spread":0.2523464184949132,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}