{"id":"W34986029","doi":"10.1089/cmb.2021.0438","title":"Exploiting Conversation Structure in Unsupervised Topic Segmentation for Emails","year":2010,"lang":"en","type":"article","venue":"Empirical Methods in Natural Language Processing","topic":"Personal Information Management and User Behavior","field":"Decision Sciences","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Latent Dirichlet allocation; Computer science; Topic model; Artificial intelligence; Natural language processing; Conversation; Segmentation; Exploit; Market segmentation; Thread (computing); Text segmentation; Information retrieval; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002732757,0.0007635469,0.0008066572,0.003993297,0.001026455,0.001350859,0.0008228426,0.001419546,0.001953487],"category_scores_gemma":[0.01166907,0.0004864434,0.0007776149,0.00164479,0.0007591173,0.002565441,0.001049733,0.001226282,0.00149955],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001072829,"about_ca_system_score_gemma":0.001091656,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004840827,"about_ca_topic_score_gemma":0.007943881,"domain_scores_codex":[0.998323,0.0007619077,0.00008100609,0.0004385739,0.0001958332,0.0001994865],"domain_scores_gemma":[0.9934228,0.004861163,0.0005407316,0.0003445669,0.0005888172,0.0002419019],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00143887,0.0006650648,0.05062482,0.0005870352,0.0002458985,0.0003909521,0.003852312,0.1353786,0.0332248,0.01890531,0.0139189,0.7407674],"study_design_scores_gemma":[0.00002076429,0.00005787202,0.01044869,0.00003213785,0.00003092186,0.0001151868,0.0003847849,0.9657826,0.004934063,0.01550183,0.002656739,0.00003441056],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3361197,0.001403931,0.6511336,0.000896129,0.000124056,0.0003041842,0.001312834,0.003110141,0.005595543],"genre_scores_gemma":[0.8661972,0.0002516952,0.1285883,0.00009874607,0.0001724036,0.0002021647,0.00203808,0.0002204974,0.002230914],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004840827,"threshold_uncertainty_score":0.01445234,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2587489947827946,"score_gpt":0.5735565170127669,"score_spread":0.3148075222299723,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}