{"id":"W4390913029","doi":"10.21437/iscslp.2008-51","title":"Subword Latent Semantic Analysis for TextTiling-based Automatic Story Segmentation of Chinese Broadcast News","year":2008,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Science North","funders":"","keywords":"Bigram; Computer science; Artificial intelligence; Speech recognition; Latent semantic analysis; Natural language processing; Robustness (evolution); Mandarin Chinese; Sentence; Treebank; Hidden Markov model; Segmentation; Character (mathematics); Text segmentation; Trigram; Annotation; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005678638,0.0006271026,0.0005203714,0.002265462,0.0004947454,0.0006928011,0.0004425704,0.0003603066,0.002144598],"category_scores_gemma":[0.001972425,0.0002067946,0.0007424265,0.00115447,0.0003969065,0.001333267,0.0005755238,0.0004705325,0.001103458],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004498801,"about_ca_system_score_gemma":0.0007044299,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002542903,"about_ca_topic_score_gemma":0.003873707,"domain_scores_codex":[0.9994602,0.000168188,0.00004489034,0.0001482407,0.0001187009,0.00005985655],"domain_scores_gemma":[0.9991366,0.0003618903,0.0001472401,0.00009089592,0.0002121303,0.00005125628],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0006807031,0.0001700283,0.004271809,0.0003482946,0.00009923321,0.0001727759,0.0006271857,0.01072782,0.1223878,0.00485887,0.002693746,0.8529617],"study_design_scores_gemma":[0.00006829746,0.0003309635,0.01260865,0.000043349,0.0001499933,0.000246225,0.0006156108,0.8747351,0.09608804,0.007803967,0.007221562,0.00008822676],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.169205,0.0007132356,0.8231844,0.0002078298,0.00008574624,0.0001441782,0.0007967732,0.003304349,0.002358538],"genre_scores_gemma":[0.6523174,0.0003502867,0.3412097,0.00008306625,0.0001160917,0.0002657953,0.002922549,0.0003191616,0.002415953],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002542903,"threshold_uncertainty_score":0.007174432,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03312502395368702,"score_gpt":0.272342365913709,"score_spread":0.239217341960022,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}