{"id":"W4414031611","doi":"10.1613/jair.1.18144","title":"Optimal Decision Trees for Interpretable and Constrained Clustering","year":2025,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"University of Toronto; Natural Sciences and Engineering Research Council of Canada; Government of Canada; Canadian Institute for Advanced Research","keywords":"Cluster analysis; Computer science; Artificial intelligence; Decision tree; Machine learning; Data mining","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00348996,0.001206785,0.001422215,0.001597161,0.001006502,0.001929714,0.001791022,0.001964373,0.004510456],"category_scores_gemma":[0.01743573,0.000644221,0.001512959,0.002563738,0.001607481,0.002737519,0.00194821,0.003010674,0.0008240878],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001787726,"about_ca_system_score_gemma":0.002291686,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002436253,"about_ca_topic_score_gemma":0.004326637,"domain_scores_codex":[0.9965425,0.001680678,0.0002090597,0.000582343,0.0007716624,0.0002136102],"domain_scores_gemma":[0.9929473,0.005130474,0.0004094226,0.0005619102,0.0007754375,0.0001754376],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001579801,0.0001163934,0.001032189,0.0003127107,0.00007154376,0.0001673352,0.000377389,0.718653,0.002099583,0.152694,0.006573679,0.1177442],"study_design_scores_gemma":[0.00002183733,0.0000281003,0.0001118471,0.00003553891,0.0000113036,0.00003728117,0.00004762426,0.8680327,0.0005904954,0.129468,0.001604278,0.00001087722],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.007387942,0.0002179588,0.9901351,0.0003168136,0.00003076136,0.00007468793,0.0002194348,0.000207021,0.001410278],"genre_scores_gemma":[0.194042,0.0003969581,0.8015698,0.0003095106,0.00008040686,0.000402727,0.001356127,0.0002140714,0.001628381],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004510456,"threshold_uncertainty_score":0.01845688,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1133541147254846,"score_gpt":0.4426594675277677,"score_spread":0.3293053528022831,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}