{"id":"W4414739115","doi":"10.22250/24107190-2025-11-3-87","title":"Corpus-based discourse analysis of connected speech phenomena in typologically diverse languages","year":2025,"lang":"en","type":"article","venue":"Theoretical and Applied Linguistics","topic":"Discourse Analysis and Cultural Communication","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Scripting language; Annotation; Python (programming language); Computational linguistics; Speech corpus; Corpus linguistics; Natural language; Text corpus","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003958081,0.0006649788,0.0005468841,0.009439089,0.001864484,0.002665513,0.001042133,0.0006854267,0.004317331],"category_scores_gemma":[0.009181015,0.0003511913,0.0006216606,0.005862522,0.001423952,0.002107766,0.002501993,0.0009147793,0.0009520021],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001268752,"about_ca_system_score_gemma":0.002123036,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002981964,"about_ca_topic_score_gemma":0.003888549,"domain_scores_codex":[0.9949636,0.002538141,0.0004479245,0.001191559,0.0007429597,0.0001157882],"domain_scores_gemma":[0.9903122,0.006313483,0.0005035007,0.001147773,0.001524683,0.000198333],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0005913936,0.0006759804,0.02016732,0.003283543,0.0003282557,0.00190183,0.06608295,0.009232857,0.1407667,0.09172229,0.0103007,0.6549463],"study_design_scores_gemma":[0.0001918333,0.0006861563,0.1216224,0.001612313,0.0005681681,0.003503358,0.07108557,0.193423,0.1630396,0.09912545,0.3446185,0.0005237925],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1948187,0.001289778,0.7713728,0.0003657627,0.0002042421,0.002693627,0.01054613,0.002064516,0.01664446],"genre_scores_gemma":[0.2874033,0.0004594429,0.6919478,0.00005792571,0.00006196276,0.004240481,0.01198652,0.0004255957,0.003416979],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009439089,"threshold_uncertainty_score":0.02093256,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01298485957063372,"score_gpt":0.339433345069315,"score_spread":0.3264484854986813,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}