{"id":"W2894762547","doi":"10.1145/3209280.3229114","title":"Improving Short Text Clustering by Similarity Matrix Sparsification","year":2018,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Cluster analysis; Similarity (geometry); Computer science; Artificial intelligence; Embedding; Word (group theory); Correlation clustering; Document clustering; Pattern recognition (psychology); Data mining; Mathematics; Image (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002515958,0.00007993499,0.00007358059,0.00003832027,0.0001181304,0.0001534998,0.0005699813,0.00004956465,0.0000356249],"category_scores_gemma":[0.00002200514,0.00007608772,0.00002388574,0.0001233903,0.00002330358,0.0004284912,0.000297373,0.00007240388,0.00007965206],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004166568,"about_ca_system_score_gemma":0.00002259518,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000152745,"about_ca_topic_score_gemma":0.00004656942,"domain_scores_codex":[0.9990968,0.00001833385,0.0001704547,0.0003472853,0.0001606687,0.0002064524],"domain_scores_gemma":[0.9992687,0.00001939292,0.0000305858,0.000560865,0.000060738,0.0000597098],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000006966916,0.00008057496,0.003068417,0.0000431836,0.00001658107,0.000003651939,0.001086361,0.0003442502,0.1202833,0.03155601,0.009344026,0.8341667],"study_design_scores_gemma":[0.00005223826,0.00002117427,0.0003008949,0.000003880256,0.00000169133,0.000004179746,0.00001400387,0.9856964,0.01087737,0.0002919545,0.002626003,0.0001101823],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01941253,0.0000278578,0.9735232,0.0005733892,0.0002839646,0.00008773535,5.052379e-7,0.0002410299,0.005849799],"genre_scores_gemma":[0.8064047,0.000001764149,0.1925221,0.0002308769,0.0001190616,0.000003560254,9.444128e-7,0.000004779789,0.0007123072],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9853522,"threshold_uncertainty_score":0.3102767,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02896945668198172,"score_gpt":0.2747629703535149,"score_spread":0.2457935136715332,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}