{"id":"W2167528246","doi":"10.1145/1099554.1099665","title":"Document clustering using character N-grams","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Cluster analysis; Computer science; Character (mathematics); Word (group theory); Preprocessor; Artificial intelligence; Robustness (evolution); Curse of dimensionality; Document clustering; Pattern recognition (psychology); Dimension (graph theory); n-gram; Natural language processing; Language model; Mathematics; Combinatorics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001402325,0.00009025334,0.00007860472,0.00006325151,0.00006496269,0.0002494866,0.0005644956,0.00003894621,0.00004465183],"category_scores_gemma":[0.000007613528,0.00007120064,0.00003046842,0.0001464403,0.00001175856,0.001000664,0.0003443756,0.00009377712,0.00004115908],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007569055,"about_ca_system_score_gemma":0.00001642257,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003148049,"about_ca_topic_score_gemma":0.00001012269,"domain_scores_codex":[0.9992743,0.00001388639,0.0001318556,0.0002178994,0.0001576566,0.0002043781],"domain_scores_gemma":[0.9995545,0.00001107444,0.0000440562,0.0003097419,0.00003302204,0.00004759743],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000003114579,0.00003776265,0.00010208,0.00001995924,0.00001030644,0.00001989153,0.0006414315,0.00006324877,0.0430152,0.05325373,0.0004395286,0.9023938],"study_design_scores_gemma":[0.0003479068,0.00005876139,0.0001144704,0.0001148953,0.000007945241,0.0002052236,0.0000152909,0.6126871,0.3387896,0.01875537,0.02817942,0.0007240427],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006738839,0.0002952531,0.9895986,0.001299876,0.0001046455,0.00007209798,8.732069e-8,0.0007978627,0.001092759],"genre_scores_gemma":[0.3345427,0.000002260651,0.6640108,0.0009543775,0.00009125765,0.000002388103,2.66518e-7,0.000004442113,0.0003915606],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9016697,"threshold_uncertainty_score":0.2903478,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01723385651549489,"score_gpt":0.2888215937922208,"score_spread":0.271587737276726,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}