{"id":"W1749331913","doi":"10.1007/978-3-642-03070-3_50","title":"New Semantic Similarity Based Model for Text Clustering Using Extended Gloss Overlaps","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; WordNet; Semantic similarity; Vector space model; Ambiguity; Cluster analysis; Natural language processing; Artificial intelligence; Explicit semantic analysis; Information retrieval; Similarity (geometry); Semantic computing; Semantic Web; Semantic technology; Image (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001827222,0.0008794483,0.001452805,0.004580338,0.001375021,0.002497948,0.00310965,0.001651171,0.003796825],"category_scores_gemma":[0.00588689,0.0005598966,0.001925755,0.005126528,0.0008085662,0.006724368,0.00256084,0.001339852,0.002511245],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00131601,"about_ca_system_score_gemma":0.001450241,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00836304,"about_ca_topic_score_gemma":0.01166058,"domain_scores_codex":[0.9966463,0.0006202864,0.0004194753,0.0007845332,0.001375473,0.0001538443],"domain_scores_gemma":[0.9975338,0.0006565401,0.0001672967,0.0006468479,0.0009198676,0.00007564802],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009212957,0.0006024571,0.003856904,0.0007403235,0.0005379464,0.0004821031,0.001148169,0.1139982,0.02611336,0.09465429,0.0245038,0.7324411],"study_design_scores_gemma":[0.00003229848,0.00008795391,0.001090516,0.00005556083,0.0001227103,0.0002631301,0.0001517946,0.9444794,0.006864402,0.03675791,0.01003093,0.00006345367],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01034052,0.0003948553,0.9837336,0.0001406826,0.00008058087,0.0001786091,0.001055482,0.001964521,0.002111153],"genre_scores_gemma":[0.2101151,0.0005527406,0.7714465,0.0001932946,0.0001761944,0.0007761515,0.007568211,0.0007491472,0.008422595],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.00836304,"threshold_uncertainty_score":0.01662874,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04332717124989038,"score_gpt":0.2861407243885287,"score_spread":0.2428135531386383,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}