{"id":"W2129546202","doi":"10.1109/tkde.2009.174","title":"An Efficient Concept-Based Mining Model for Enhancing Text Clustering","year":2009,"lang":"en","type":"article","venue":"IEEE Transactions on Knowledge and Data Engineering","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":131,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Sentence; Natural language processing; Phrase; Term (time); Similarity (geometry); Semantics (computer science); Cluster analysis; Artificial intelligence; Document clustering; Meaning (existential); Word (group theory); Measure (data warehouse); Information retrieval; Similarity measure; Semantic similarity; Data mining; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002401253,0.00120815,0.00156362,0.002691224,0.001174269,0.001540669,0.003165118,0.001481713,0.001851873],"category_scores_gemma":[0.005989402,0.000546024,0.0016378,0.003836383,0.0006500463,0.003774294,0.001281214,0.001601557,0.001382491],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00146753,"about_ca_system_score_gemma":0.002304716,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006205599,"about_ca_topic_score_gemma":0.006492814,"domain_scores_codex":[0.998032,0.0004218387,0.0001275843,0.0004775755,0.0008587477,0.00008221613],"domain_scores_gemma":[0.9980314,0.0007630727,0.0001266179,0.0001815688,0.0008494802,0.00004779686],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002577812,0.0004638395,0.002256577,0.0004817782,0.0002271726,0.0003603792,0.0006281056,0.3258717,0.01287088,0.05980281,0.01117904,0.5855999],"study_design_scores_gemma":[0.00001140514,0.00002872181,0.0001354591,0.00001142023,0.00001541556,0.000096718,0.00002659438,0.9864913,0.001324792,0.009396266,0.002448864,0.00001302773],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.004177923,0.0001679203,0.9939487,0.00014104,0.00003156213,0.0001640004,0.0001361474,0.0005482856,0.0006844412],"genre_scores_gemma":[0.06240765,0.0003027297,0.9339467,0.0001623432,0.00004591308,0.0005506253,0.0007043881,0.00008416014,0.001795447],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006205599,"threshold_uncertainty_score":0.01269919,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02779020595604768,"score_gpt":0.2810140212831417,"score_spread":0.253223815327094,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}