{"id":"W2309226206","doi":"","title":"Scalable clustering of categorical data and applications","year":2004,"lang":"en","type":"article","venue":"","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Cluster analysis; Data mining; Categorical variable; Tuple; Redundancy (engineering); Scalability; Constrained clustering; Correlation clustering; Canopy clustering algorithm; Artificial intelligence; Machine learning; Database; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006070415,0.001354936,0.002100018,0.008191299,0.002363347,0.004164258,0.003860817,0.001939039,0.002911809],"category_scores_gemma":[0.03205987,0.0009515692,0.002432752,0.01257941,0.001844476,0.005008186,0.005109983,0.002551059,0.001906108],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003201118,"about_ca_system_score_gemma":0.003985344,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007187844,"about_ca_topic_score_gemma":0.007564934,"domain_scores_codex":[0.9924213,0.002240857,0.000609577,0.001613298,0.002684601,0.0004303097],"domain_scores_gemma":[0.9817289,0.007504489,0.00158597,0.004769325,0.003813914,0.0005974278],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004281297,0.0002445713,0.01059619,0.0009858909,0.00031338,0.0003028757,0.001347047,0.2419793,0.007583478,0.12482,0.02923563,0.5821635],"study_design_scores_gemma":[0.00004152129,0.0001028886,0.00306772,0.0001041733,0.00004914991,0.0002230053,0.0005345994,0.7432346,0.004777545,0.2271723,0.02058541,0.0001070159],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.008386308,0.0008248515,0.9838801,0.0006077638,0.0000773105,0.0002286346,0.001103442,0.003263848,0.001627687],"genre_scores_gemma":[0.1277333,0.0008293709,0.8629026,0.0002806993,0.0001739239,0.0006501065,0.005011114,0.0005506512,0.001868154],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008191299,"threshold_uncertainty_score":0.03210384,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03283988717391283,"score_gpt":0.2669851235728413,"score_spread":0.2341452363989284,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}