{"id":"W133424112","doi":"10.1007/978-3-642-30353-1_29","title":"Text Similarity Using Google Tri-grams","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Topic Modeling","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Similarity (geometry); Artificial intelligence; Set (abstract data type); Natural language processing; Word (group theory); n-gram; Data set; Information retrieval; Language model; Mathematics; Image (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009729846,0.001243138,0.001687332,0.0171304,0.001198744,0.002829449,0.00108541,0.001320281,0.01576247],"category_scores_gemma":[0.006178006,0.0003891061,0.001676563,0.01299483,0.0003445666,0.004489659,0.002138093,0.0008303757,0.01510495],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005947514,"about_ca_system_score_gemma":0.0009227031,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003234067,"about_ca_topic_score_gemma":0.005365049,"domain_scores_codex":[0.9977493,0.0002799762,0.0003002913,0.0004812791,0.00102528,0.0001638831],"domain_scores_gemma":[0.9974195,0.0006572529,0.0002639217,0.0003998726,0.00109739,0.0001620337],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000806107,0.0002233265,0.004912147,0.0009114697,0.0002900676,0.0002574255,0.0003676086,0.003986952,0.02029983,0.006310122,0.05811567,0.9035193],"study_design_scores_gemma":[0.0003245409,0.001156506,0.03906624,0.0004153718,0.0007221529,0.004906244,0.002517338,0.6598907,0.06802795,0.07146294,0.151127,0.0003829709],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.171812,0.01114878,0.66584,0.0009222667,0.002761372,0.001580142,0.03639806,0.06532615,0.04421132],"genre_scores_gemma":[0.4678721,0.002475245,0.4484072,0.0001810927,0.001188664,0.0007658983,0.04975494,0.003106999,0.02624795],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0171304,"threshold_uncertainty_score":0.05273074,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04910519520628955,"score_gpt":0.2708755699584219,"score_spread":0.2217703747521324,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}