{"id":"W1988047293","doi":"10.1007/s10115-013-0617-y","title":"Analyzing topics and authors in chat logs for crime investigation","year":2013,"lang":"en","type":"article","venue":"Knowledge and Information Systems","topic":"Topic Modeling","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":false,"ca_institutions":"Concordia University","funders":"Fonds Québécois de la Recherche sur la Nature et les Technologies","keywords":"Computer science; Exploit; Latent Dirichlet allocation; Online chat; Process (computing); The Internet; Order (exchange); Volume (thermodynamics); Channel (broadcasting); Topic model; Data science; World Wide Web; Information retrieval; Computer security","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00337138,0.0005706542,0.0006846928,0.00853906,0.001234479,0.002289474,0.0006488718,0.001090262,0.00137459],"category_scores_gemma":[0.02525412,0.0003115477,0.0007438794,0.005307045,0.0004386414,0.002834896,0.001024492,0.001213893,0.001217929],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007007926,"about_ca_system_score_gemma":0.001041494,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004581365,"about_ca_topic_score_gemma":0.008501317,"domain_scores_codex":[0.9966475,0.001542153,0.0003011955,0.000579159,0.0006717679,0.000258149],"domain_scores_gemma":[0.9528159,0.0371201,0.003376713,0.002141,0.003184944,0.00136139],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001425927,0.001131429,0.7873638,0.0007868768,0.0004762555,0.0006940234,0.01480506,0.007936661,0.01309652,0.003106773,0.01394983,0.1552269],"study_design_scores_gemma":[0.00006701159,0.000642915,0.6484546,0.0002584432,0.0006313599,0.001745306,0.01532379,0.2959142,0.01247077,0.008259748,0.01605828,0.0001735624],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.967697,0.000822051,0.02114632,0.0004883923,0.0000977072,0.0001552131,0.006482151,0.000922591,0.002188618],"genre_scores_gemma":[0.9831352,0.0002792429,0.00941956,0.00003730364,0.000111577,0.0001487847,0.005500834,0.00007692893,0.001290584],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.00853906,"threshold_uncertainty_score":0.01782972,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02670907494937454,"score_gpt":0.2508178429459265,"score_spread":0.224108767996552,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}