{"id":"W2167528246","doi":"10.1145/1099554.1099665","title":"Document clustering using character N-grams","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Cluster analysis; Computer science; Character (mathematics); Word (group theory); Preprocessor; Artificial intelligence; Robustness (evolution); Curse of dimensionality; Document clustering; Pattern recognition (psychology); Dimension (graph theory); n-gram; Natural language processing; Language model; Mathematics; Combinatorics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000566967,0.0009284326,0.001141594,0.004432191,0.0009170509,0.001288704,0.001108486,0.0008070644,0.001294986],"category_scores_gemma":[0.003020956,0.0003271508,0.000928005,0.006107809,0.0005816671,0.001673035,0.0009218433,0.0007812987,0.002319609],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006314492,"about_ca_system_score_gemma":0.0009741121,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002447901,"about_ca_topic_score_gemma":0.00396837,"domain_scores_codex":[0.9986005,0.000233578,0.0001288855,0.0003866234,0.0005767543,0.00007361974],"domain_scores_gemma":[0.9983565,0.000343359,0.0002568074,0.0003664735,0.000611115,0.00006562546],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001850746,0.0001696095,0.002690005,0.0003718207,0.0001785615,0.000129838,0.000291735,0.01889518,0.03858729,0.01185981,0.007051804,0.9195893],"study_design_scores_gemma":[0.00008063859,0.0003730618,0.006745408,0.0001269059,0.0001795664,0.001468695,0.0002958025,0.7858903,0.08166251,0.05678329,0.06610778,0.0002860618],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01136768,0.000739899,0.9841412,0.0001075102,0.0001258007,0.0001282221,0.0003301268,0.001880349,0.001179201],"genre_scores_gemma":[0.0667794,0.0004953223,0.9280775,0.00008210391,0.0001431287,0.0001797885,0.001208414,0.000202628,0.002831664],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004432191,"threshold_uncertainty_score":0.004867315,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01723385651549489,"score_gpt":0.2888215937922208,"score_spread":0.271587737276726,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}