{"id":"W2051195289","doi":"10.1145/1077501.1077517","title":"Clustering mixed numerical and low quality categorical data","year":2005,"lang":"en","type":"article","venue":"","topic":"Gene expression and cancer classification","field":"Biochemistry, Genetics and Molecular Biology","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Cluster analysis; Categorical variable; Correctness; Data mining; Computer science; Gene ontology; Artificial intelligence; Algorithm; Machine learning; Gene","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001267594,0.00006556296,0.00006526225,0.00001123072,0.00003936314,0.00001912243,0.0001674623,0.00006666115,0.00003609053],"category_scores_gemma":[0.00003602494,0.00005506563,0.0000142183,0.0000349227,0.00002430502,0.000004023288,0.000257254,0.0000392057,0.000009888298],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000006233149,"about_ca_system_score_gemma":0.00002387161,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001519137,"about_ca_topic_score_gemma":0.00003150424,"domain_scores_codex":[0.9993483,0.0000404443,0.0001260803,0.0003100328,0.00007148781,0.0001036332],"domain_scores_gemma":[0.9993993,0.000004909959,0.00002984756,0.000473913,0.00001846953,0.00007354388],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00006073583,0.00007744387,0.001432855,0.00001475194,0.00001490279,3.660031e-7,0.00002241601,0.00004544547,0.8600534,0.0002538011,0.03999192,0.09803198],"study_design_scores_gemma":[0.001004115,0.00009168374,0.03046998,0.000005675568,0.00001284427,0.00002391972,0.0001957261,0.02074357,0.2142436,0.00004869141,0.7327508,0.0004093825],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5487739,0.001318689,0.4366063,0.004641907,0.0002834002,0.0001965122,0.00002251635,0.00005355142,0.008103198],"genre_scores_gemma":[0.9952524,0.0001233569,0.002659678,0.0004662381,0.0002282009,0.000005439068,0.0001485774,0.000006678264,0.001109486],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6927589,"threshold_uncertainty_score":0.2245511,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05042977435946126,"score_gpt":0.3367815668567667,"score_spread":0.2863517924973054,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}