{"id":"W1982030926","doi":"10.2202/1544-6115.1261","title":"Estimating Number of Clusters Based on a General Similarity Matrix with Application to Microarray Data","year":2008,"lang":"en","type":"article","venue":"Statistical Applications in Genetics and Molecular Biology","topic":"Gene expression and cancer classification","field":"Biochemistry, Genetics and Molecular Biology","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Cluster analysis; Similarity (geometry); Data mining; Computer science; Selection (genetic algorithm); Set (abstract data type); A priori and a posteriori; Data set; Model selection; Matrix (chemical analysis); Determining the number of clusters in a data set; Artificial intelligence; Correlation clustering; CURE data clustering algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001180994,0.0001485715,0.0001630071,0.00006857177,0.00007099871,0.000006717352,0.0002775728,0.0001210866,0.000006132022],"category_scores_gemma":[0.00004170742,0.0001329145,0.00001634307,0.0001889152,0.0002026672,0.000001624603,0.0001355605,0.00007977006,0.000002760812],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001194965,"about_ca_system_score_gemma":0.0000905788,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002366261,"about_ca_topic_score_gemma":0.00001642099,"domain_scores_codex":[0.9987733,0.00006847914,0.0002649145,0.0006124662,0.00009697716,0.0001838984],"domain_scores_gemma":[0.998916,0.00003119444,0.00008692467,0.0007834006,0.00007806611,0.0001043623],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002328143,0.0003229432,0.03157435,0.00005484572,0.00003031928,0.000002396334,0.00004255694,0.009230218,0.9351785,0.01083313,0.0006240071,0.01187397],"study_design_scores_gemma":[0.006017107,0.002666916,0.07953744,0.0001072503,0.0001884588,0.0001182616,0.0001741865,0.412589,0.3810627,0.008609525,0.1066371,0.002291972],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1991773,0.00007228394,0.7997012,0.0001831674,0.00001490087,0.0004341111,0.0002193211,0.000005597769,0.0001921607],"genre_scores_gemma":[0.6997311,0.00002632785,0.2987033,0.0003224735,0.00002923393,0.0001818122,0.000977983,0.00001480094,0.00001301632],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.5541157,"threshold_uncertainty_score":0.5420097,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01388867764490385,"score_gpt":0.3370596504591929,"score_spread":0.323170972814289,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}