{"id":"W4244748443","doi":"10.31234/osf.io/xp6k2","title":"Automatic word count estimation from daylong child-centered recordings in various language environments using language-independent syllabification of speech","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Speech and dialogue systems","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Manitoba","funders":"","keywords":"Computer science; Syllable; Word (group theory); Speech recognition; Syllabification; Phonotactics; Natural language processing; Artificial intelligence; Mathematics; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0006264242,0.0003549316,0.0006160448,0.0003711806,0.00004120204,0.0002071999,0.001130462,0.0003447055,0.00009119381],"category_scores_gemma":[0.0001084858,0.0003573556,0.0001250011,0.0002263544,0.00002641686,0.0003249956,0.0008374203,0.000402589,0.0001292871],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005937069,"about_ca_system_score_gemma":0.0001303782,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01622307,"about_ca_topic_score_gemma":0.0004599797,"domain_scores_codex":[0.9969718,0.0002005482,0.0009017131,0.0008773932,0.0007110393,0.0003374914],"domain_scores_gemma":[0.9974118,0.0001257303,0.0008208457,0.00152426,0.00002990974,0.0000874236],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001274795,0.001520067,0.03642093,0.001937007,0.0006550951,0.0004320501,0.05773976,0.09345058,0.07903863,0.0008317118,0.0002036558,0.727643],"study_design_scores_gemma":[0.0008848088,0.00002885571,0.02952866,0.00118519,0.00003571716,0.00002269478,0.000469631,0.9541683,0.01255878,0.0006424081,0.00001111256,0.0004638636],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5942411,0.0002756975,0.4029953,0.00004651218,0.0009818546,0.000804906,0.0000397223,0.00008484508,0.0005299951],"genre_scores_gemma":[0.9076674,0.00001506833,0.09176679,0.00003832933,0.00009184281,0.00002301603,0.0003154036,0.00002698008,0.00005520279],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8607177,"threshold_uncertainty_score":0.9998878,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01606227370194699,"score_gpt":0.2556369938349796,"score_spread":0.2395747201330326,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}