{"id":"W1982030926","doi":"10.2202/1544-6115.1261","title":"Estimating Number of Clusters Based on a General Similarity Matrix with Application to Microarray Data","year":2008,"lang":"en","type":"article","venue":"Statistical Applications in Genetics and Molecular Biology","topic":"Gene expression and cancer classification","field":"Biochemistry, Genetics and Molecular Biology","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Cluster analysis; Similarity (geometry); Data mining; Computer science; Selection (genetic algorithm); Set (abstract data type); A priori and a posteriori; Data set; Model selection; Matrix (chemical analysis); Determining the number of clusters in a data set; Artificial intelligence; Correlation clustering; CURE data clustering algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006226114,0.0007892174,0.001406398,0.003008774,0.001179026,0.001379793,0.002053657,0.001604839,0.0008057265],"category_scores_gemma":[0.02476651,0.0006132261,0.001211646,0.003907684,0.001522078,0.002139454,0.001814584,0.001427045,0.000409177],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001461457,"about_ca_system_score_gemma":0.001423962,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005072501,"about_ca_topic_score_gemma":0.005334629,"domain_scores_codex":[0.9967937,0.001394783,0.0001673312,0.0007583108,0.0007464759,0.0001393791],"domain_scores_gemma":[0.9870023,0.008862638,0.0009891302,0.00122371,0.001631618,0.000290641],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002220669,0.00009842853,0.007625191,0.0002301796,0.0001977909,0.0001697523,0.000348329,0.7888135,0.008715658,0.03757438,0.001755477,0.1542493],"study_design_scores_gemma":[0.000009236901,0.00002552327,0.001074493,0.000007225554,0.00001095692,0.00006426636,0.00002362949,0.9805481,0.0008333155,0.01696982,0.0004139088,0.00001958479],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01374941,0.000150027,0.985536,0.00008867662,0.0000109787,0.00005198904,0.00004501539,0.000175842,0.0001920302],"genre_scores_gemma":[0.205387,0.0004446079,0.7925681,0.00005926183,0.00007822428,0.000201363,0.0003632726,0.0001020596,0.000796117],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006226114,"threshold_uncertainty_score":0.03292722,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01388867764490385,"score_gpt":0.3370596504591929,"score_spread":0.323170972814289,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}