{"id":"W2140920152","doi":"10.1007/978-3-540-68825-9_27","title":"A Statistical Model for Topic Segmentation and Clustering","year":2008,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Topic Modeling","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Cluster analysis; Segmentation; Document clustering; Artificial intelligence; Statistical model; Brown clustering; Hierarchical clustering; Topic model; Bayesian probability; Natural language processing; Pattern recognition (psychology); Data mining; Fuzzy clustering; Canopy clustering algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005336261,0.001342161,0.002315025,0.00342689,0.001524252,0.003510129,0.00498053,0.00323211,0.005505194],"category_scores_gemma":[0.01682482,0.001584598,0.003188261,0.005929234,0.001578692,0.005286216,0.002320199,0.003527618,0.004366418],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002196394,"about_ca_system_score_gemma":0.002076265,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01015524,"about_ca_topic_score_gemma":0.01279186,"domain_scores_codex":[0.9966348,0.001339731,0.0002448637,0.0008920634,0.0006476075,0.0002409501],"domain_scores_gemma":[0.9906703,0.006707581,0.0004251587,0.001078736,0.0009019264,0.000216334],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003416312,0.0001954781,0.002276758,0.0004635766,0.0003758618,0.0002338087,0.0008905799,0.3143768,0.005269645,0.2846902,0.02459793,0.3662879],"study_design_scores_gemma":[0.00002055683,0.0000234744,0.0004107228,0.00002569793,0.00004810661,0.0001044223,0.0000352759,0.8550005,0.00064882,0.139152,0.004498846,0.00003171243],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001945966,0.0004412885,0.9958367,0.0001893182,0.00005498687,0.00004602618,0.0002905944,0.0006152461,0.0005799215],"genre_scores_gemma":[0.1377978,0.001909669,0.8402444,0.0003471058,0.0007115449,0.001069882,0.004497563,0.001115098,0.01230697],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01015524,"threshold_uncertainty_score":0.02822113,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0366026267238402,"score_gpt":0.2742441229591897,"score_spread":0.2376414962353495,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}