{"id":"W4403794801","doi":"10.48550/arxiv.2409.17353","title":"Internalizing ASR with Implicit Chain of Thought for Efficient Speech-to-Speech Conversational LLM","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Speech and dialogue systems","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; University of British Columbia","keywords":"Speech recognition; Psychology; Indirect speech; Computer science; Linguistics; Cognitive psychology; Natural language processing; Philosophy","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":{"n_in":0,"stratum":"fund_new","weight":1678.9,"opus":{"tier":"OUT","genre":"empirical","about_ca":false,"confidence":"high","reason":"Machine learning method internalizing ASR chain of thought in a speech-to-speech conversational LLM; a model architecture contribution."},"gpt":{"tier":"OUT","genre":"empirical","about_ca":false,"confidence":"high","reason":"The preprint develops a speech-to-speech language-model method, not a study of research practice."},"grok":{"tier":"OUT","genre":"empirical","about_ca":false,"confidence":"high","reason":"Speech LLM engineering for ASR chain-of-thought; AI systems research, not metaresearch."}},"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002124523,0.002162276,0.00113822,0.0008512423,0.0005566685,0.002066953,0.001766903,0.001529622,0.006241773],"category_scores_gemma":[0.007237066,0.0006613979,0.001510157,0.0005326312,0.001022812,0.002998732,0.003057613,0.00347602,0.006021843],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008674146,"about_ca_system_score_gemma":0.001784994,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003272189,"about_ca_topic_score_gemma":0.005488124,"domain_scores_codex":[0.9984559,0.0005547066,0.000098508,0.0005311866,0.0002168422,0.0001427638],"domain_scores_gemma":[0.9977481,0.001083072,0.000126212,0.0005641638,0.0003163877,0.0001621257],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001310193,0.0005553496,0.00349463,0.0004697012,0.0002993763,0.0004557413,0.001097743,0.1342251,0.0686627,0.01350202,0.02506351,0.750864],"study_design_scores_gemma":[0.0000639509,0.0001324622,0.0004548039,0.00002693716,0.00006055686,0.0001283209,0.0001016346,0.9660271,0.01474314,0.0119071,0.006311247,0.00004282755],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02784811,0.000901688,0.943743,0.0005987083,0.0003042197,0.0001660774,0.000776912,0.02215293,0.003508311],"genre_scores_gemma":[0.5984261,0.0004976974,0.3841808,0.0007312722,0.0003773558,0.0004610787,0.003459199,0.001725788,0.0101406],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006241773,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04715818690535335,"score_gpt":0.2061998562407613,"score_spread":0.159041669335408,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}