{"id":"W2131912994","doi":"10.1109/icdew.2007.4401066","title":"Document Representation and Dimension Reduction for Text Clustering","year":2007,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Cluster analysis; Dimensionality reduction; Computer science; Document clustering; Artificial intelligence; Pattern recognition (psychology); Representation (politics); Context (archaeology); Benchmark (surveying); Word (group theory); Dimension (graph theory); Natural language processing; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00209821,0.0009500743,0.001277553,0.0043619,0.0009551603,0.001797058,0.001006586,0.0007763233,0.002335153],"category_scores_gemma":[0.009199312,0.0002691926,0.001228747,0.006258083,0.0004582229,0.001467797,0.0008825315,0.001028969,0.002324112],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009154171,"about_ca_system_score_gemma":0.001340195,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002438364,"about_ca_topic_score_gemma":0.002241694,"domain_scores_codex":[0.9976168,0.0009675795,0.0002387008,0.0003712818,0.0007036518,0.0001019902],"domain_scores_gemma":[0.9972253,0.001115284,0.0002260035,0.0005962054,0.0007885131,0.00004875025],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001942384,0.0001513128,0.001232644,0.0004834591,0.0001570737,0.00008418165,0.0003218447,0.03769486,0.01159173,0.01660562,0.01646832,0.9150147],"study_design_scores_gemma":[0.0001077144,0.0002868542,0.005146157,0.0001462612,0.0002045409,0.0006515923,0.0004957599,0.857314,0.02451032,0.06671607,0.04425116,0.0001696028],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0158168,0.001844474,0.975069,0.0005475485,0.000178619,0.0003578377,0.001513622,0.00266266,0.002009549],"genre_scores_gemma":[0.07137626,0.0009033469,0.9219764,0.00008674904,0.0001702403,0.0006139782,0.003092804,0.0001503311,0.001629881],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0043619,"threshold_uncertainty_score":0.01109648,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02785592813353458,"score_gpt":0.3078267137122061,"score_spread":0.2799707855786715,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}