{"id":"W1633328346","doi":"10.1007/978-3-642-21043-3_26","title":"Comparison of Semantic Similarity for Different Languages Using the Google n-gram Corpus and Second-Order Co-occurrence Measures","year":2011,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Topic Modeling","field":"Computer Science","cited_by":50,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Natural language processing; Similarity (geometry); German; Semantic similarity; Artificial intelligence; Word (group theory); n-gram; Word order; Language model; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001406088,0.001016937,0.000928979,0.01110527,0.00141258,0.002344336,0.0006835499,0.0008859193,0.003371337],"category_scores_gemma":[0.00996111,0.0002495293,0.00109541,0.01062219,0.0006393766,0.003278265,0.001517776,0.0009415905,0.002379413],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007516271,"about_ca_system_score_gemma":0.001266467,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008607333,"about_ca_topic_score_gemma":0.0150723,"domain_scores_codex":[0.9978199,0.0006733983,0.0003038181,0.0003390438,0.0007038592,0.0001600548],"domain_scores_gemma":[0.9939115,0.003502298,0.0002420354,0.000492137,0.001595745,0.0002562882],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.007760614,0.001363785,0.09230053,0.008341644,0.001604479,0.002478025,0.006824027,0.01492691,0.08588388,0.01600226,0.09216452,0.6703493],"study_design_scores_gemma":[0.0005808832,0.001942027,0.3958008,0.001136577,0.00180328,0.006863606,0.01847528,0.3076826,0.09425928,0.02573969,0.1448314,0.0008846064],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.912475,0.003305301,0.02182885,0.0003358932,0.0003809066,0.0002386999,0.04162709,0.003872497,0.01593574],"genre_scores_gemma":[0.815459,0.001333271,0.05328684,0.00007514988,0.0001091292,0.0003507056,0.1255648,0.000872025,0.002948952],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01110527,"threshold_uncertainty_score":0.01711446,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07990233339630563,"score_gpt":0.3274145852463676,"score_spread":0.247512251850062,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}