{"id":"W4402980134","doi":"10.1109/otcon60325.2024.10687796","title":"Merging Speech Recognition Capabilities with Large Language Models for Enhanced Communication and Interaction","year":2024,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Horizon College and Seminary","funders":"","keywords":"Computer science; Speech recognition; Natural language processing; Language model; Human–computer interaction; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002630567,0.00007943656,0.00008247961,0.0001116398,0.00009555095,0.0002659364,0.0001177981,0.00003237748,0.00007477409],"category_scores_gemma":[0.0000283468,0.00006329503,0.00002924545,0.0001292847,0.00001926281,0.001187416,0.00004106144,0.00006774874,0.0000195178],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000302382,"about_ca_system_score_gemma":0.00001769787,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005001737,"about_ca_topic_score_gemma":0.0001492404,"domain_scores_codex":[0.999406,0.00004207457,0.0001173278,0.0002230168,0.00009028609,0.000121228],"domain_scores_gemma":[0.9993466,0.0003163815,0.00002460928,0.0002051866,0.00007532303,0.00003190874],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001911893,0.00003503994,0.000001569643,0.0000953906,0.00003010256,0.000002271919,0.004981036,0.000005894355,0.005250083,0.02442213,0.0002870365,0.9648703],"study_design_scores_gemma":[0.0003900276,0.00009943294,0.00001695107,0.0003220903,0.00002481556,0.00005892601,0.006244436,0.6707386,0.2774923,0.04272028,0.001617664,0.0002744737],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1104484,0.0002288002,0.8727171,0.0008205027,0.00009084825,0.0002272854,0.0000106319,0.0003370568,0.01511932],"genre_scores_gemma":[0.7934752,0.0000684674,0.2056908,0.0001238507,0.00001674456,0.00007554184,0.00002080182,0.000007945784,0.0005206778],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9645959,"threshold_uncertainty_score":0.2581097,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02777832604551551,"score_gpt":0.2801247445404287,"score_spread":0.2523464184949132,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}