{"id":"W2772904413","doi":"10.1109/tnnls.2017.2770167","title":"An Algorithm for Clustering Categorical Data With Set-Valued Features","year":2017,"lang":"en","type":"article","venue":"IEEE Transactions on Neural Networks and Learning Systems","topic":"Advanced Clustering Algorithms Research","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Science Foundation of Shanxi Province; Simon Fraser University; Kungliga Tekniska Högskolan; Shanxi Scholarship Council of China; National Natural Science Foundation of China","keywords":"Cluster analysis; Computer science; Categorical variable; Data mining; Initialization; Algorithm; Set (abstract data type); CURE data clustering algorithm; Canopy clustering algorithm; Data set; Determining the number of clusters in a data set; Data stream clustering; Heuristic; Correlation clustering; Pattern recognition (psychology); Artificial intelligence; Machine learning","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002343394,0.001007942,0.001203255,0.003276791,0.001428751,0.001349667,0.002924953,0.001562478,0.001966655],"category_scores_gemma":[0.007632503,0.0006180292,0.001357549,0.004192575,0.0009660269,0.001932651,0.001888903,0.001934475,0.001293916],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001202808,"about_ca_system_score_gemma":0.002249151,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005731948,"about_ca_topic_score_gemma":0.006185837,"domain_scores_codex":[0.9980597,0.0004785909,0.0001874502,0.0005148196,0.0006545649,0.0001047456],"domain_scores_gemma":[0.9975873,0.0008276684,0.0002265738,0.0003346443,0.0009291058,0.00009469551],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002852362,0.0001314407,0.003544597,0.0001925043,0.0001560286,0.0001145346,0.00054138,0.1414619,0.006703022,0.02795702,0.009582253,0.8093301],"study_design_scores_gemma":[0.00006906468,0.0001019408,0.001088106,0.00003859586,0.00002901592,0.000275578,0.0001527266,0.9536994,0.004045152,0.03313155,0.007314512,0.00005440278],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002587914,0.00007127575,0.996289,0.00006382057,0.00002912724,0.00009949491,0.00008512141,0.00050792,0.0002662652],"genre_scores_gemma":[0.02855809,0.00006009482,0.9700918,0.0000508367,0.00002102986,0.000281941,0.0003301852,0.00006760316,0.0005383992],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005731948,"threshold_uncertainty_score":0.01239324,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04585140131797229,"score_gpt":0.3288737313040195,"score_spread":0.2830223299860473,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}