{"id":"W7154582775","doi":"10.48448/nyzm-xp21","title":"Statistical Word Segmentation in Unfamiliar Speech","year":2025,"lang":"","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Pupillometry; Statistical learning; Flexibility (engineering); Segmentation; Word (group theory); Text segmentation; Statistical analysis; Speech segmentation; Statistical model; Phonetics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.005121083,0.001269484,0.001254465,0.006700224,0.0005846533,0.0007932989,0.002813736,0.000680921,0.01603452],"category_scores_gemma":[0.002423014,0.001366675,0.0001142538,0.01210664,0.007340546,0.0008966193,0.0010212,0.001657923,0.02343219],"about_ca_system_candidate":true,"about_ca_system_consensus":true,"about_ca_system_score_codex":0.004471599,"about_ca_system_score_gemma":0.01134937,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002305848,"about_ca_topic_score_gemma":0.006762929,"domain_scores_codex":[0.9876655,0.000552505,0.001968507,0.003447983,0.003893598,0.002471903],"domain_scores_gemma":[0.9953956,0.0007349513,0.0008298592,0.001650031,0.0007040303,0.0006855478],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003702875,0.003127417,0.00876631,0.0006512158,0.0001628475,0.0007635865,0.001081228,0.002293126,0.02891927,0.106046,0.0551868,0.7926319],"study_design_scores_gemma":[0.01990462,0.001931443,0.06607416,0.01094424,0.001403566,0.0001829343,0.01178932,0.527531,0.01660844,0.08018736,0.25031,0.01313296],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.004569797,0.000697716,0.08691876,0.0008714527,0.005885901,0.005542646,0.003386556,0.0005454823,0.8915817],"genre_scores_gemma":[0.1280067,0.001153865,0.4212747,0.001495785,0.001071678,0.0001860363,0.001262005,0.001394766,0.4441544],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.7794989,"threshold_uncertainty_score":0.9993501,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01879734063685194,"score_gpt":0.3309659688284309,"score_spread":0.3121686281915789,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}