{"id":"W7154582775","doi":"10.48448/nyzm-xp21","title":"Statistical Word Segmentation in Unfamiliar Speech","year":2025,"lang":"","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Pupillometry; Statistical learning; Flexibility (engineering); Segmentation; Word (group theory); Text segmentation; Statistical analysis; Speech segmentation; Statistical model; Phonetics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003205738,0.0002196767,0.0001066948,0.0001739034,0.0001111671,0.0005144317,0.0001579473,0.0001877584,0.001845682],"category_scores_gemma":[0.001958493,0.00009175559,0.0000781983,0.00008526727,0.0003414423,0.0006300117,0.0004243434,0.0001469172,0.0003796483],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002009212,"about_ca_system_score_gemma":0.0002013078,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001118049,"about_ca_topic_score_gemma":0.001736266,"domain_scores_codex":[0.9998097,0.0000499834,0.00001462196,0.00006529236,0.0000396172,0.0000208713],"domain_scores_gemma":[0.9992374,0.0003205253,0.0001573294,0.0001005106,0.0001224667,0.00006176964],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004238524,0.00006450763,0.03328649,0.0001495649,0.00001544844,0.0004424303,0.001610277,0.002607488,0.7897802,0.001136565,0.0004003252,0.1700829],"study_design_scores_gemma":[0.00003954747,0.001595609,0.57863,0.00006765275,0.00005278387,0.001532449,0.00306091,0.08714052,0.3171561,0.004760706,0.005904806,0.00005889354],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9867703,0.00008229368,0.01047407,0.00003453611,0.000007796221,0.00001477634,0.0000407796,0.00009663856,0.00247888],"genre_scores_gemma":[0.9959227,0.00003593016,0.003330851,0.0000107505,0.000002438388,0.000008485653,0.00004345079,0.00001258793,0.0006326337],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.001845682,"threshold_uncertainty_score":0.006174445,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01879734063685194,"score_gpt":0.3309659688284309,"score_spread":0.3121686281915789,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}