{"id":"W2132052677","doi":"10.14569/ijacsa.2015.060121","title":"A Survey of Topic Modeling in Text Mining","year":2015,"lang":"en","type":"article","venue":"International Journal of Advanced Computer Science and Applications","topic":"Topic Modeling","field":"Computer Science","cited_by":368,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"Latent Dirichlet allocation; Topic model; Computer science; Probabilistic latent semantic analysis; Natural language processing; Artificial intelligence; Latent semantic analysis; Field (mathematics); Probabilistic logic; Information retrieval","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007165624,0.002088693,0.003745923,0.009564827,0.001374747,0.004509262,0.003041179,0.002203124,0.002771244],"category_scores_gemma":[0.01681403,0.001105343,0.003300958,0.01821085,0.0009602183,0.008731122,0.001771802,0.002651416,0.002589453],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001565168,"about_ca_system_score_gemma":0.002375254,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004097731,"about_ca_topic_score_gemma":0.003425356,"domain_scores_codex":[0.9935291,0.002576702,0.0008018898,0.001331183,0.00157328,0.0001877044],"domain_scores_gemma":[0.9888678,0.008538554,0.0004837128,0.0007989721,0.001146201,0.000164848],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001315607,0.0002591586,0.008975076,0.004818047,0.0006170789,0.0002799109,0.0007470395,0.0223622,0.001581075,0.0481176,0.02942097,0.8826904],"study_design_scores_gemma":[0.00009492281,0.0002607831,0.01068683,0.00261626,0.0006839056,0.00203104,0.001393651,0.4482298,0.004080579,0.2384188,0.2912305,0.0002729779],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.006272367,0.182315,0.7961861,0.004210368,0.0008489597,0.0004448869,0.002055285,0.001336062,0.006331056],"genre_scores_gemma":[0.1228998,0.3139892,0.53741,0.001826519,0.007445916,0.001764888,0.008256959,0.0006603642,0.005746305],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.009564827,"threshold_uncertainty_score":0.03789592,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0669373198898449,"score_gpt":0.3294273301311625,"score_spread":0.2624900102413176,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}