{"id":"W2478941700","doi":"10.1075/cilt.309.18isl","title":"Semantic similarity of short texts","year":2009,"lang":"en","type":"article","venue":"Amsterdam studies in the theory and history of linguistic science. Series 4, Current issues in linguistic theory","topic":"Topic Modeling","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Semantic similarity; Natural language processing; Similarity (geometry); Artificial intelligence; Word (group theory); Sentence; Variety (cybernetics); Focus (optics); Representation (politics); Matching (statistics); Longest common subsequence problem; Information retrieval; String metric; Pattern matching; String searching algorithm; Mathematics; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002225308,0.0005744817,0.001006075,0.01387948,0.001001045,0.002832079,0.001041629,0.001069732,0.004419663],"category_scores_gemma":[0.02617136,0.0002620564,0.0008436627,0.008815708,0.00127458,0.006331483,0.00197032,0.0008019678,0.001341773],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009226873,"about_ca_system_score_gemma":0.000890873,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008503963,"about_ca_topic_score_gemma":0.0008031192,"domain_scores_codex":[0.9949091,0.001258172,0.0007212418,0.0009807007,0.002000473,0.0001303883],"domain_scores_gemma":[0.9892308,0.005699278,0.00144023,0.001053247,0.00228591,0.0002906069],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001096304,0.0002685451,0.01992567,0.002661839,0.0006526271,0.0009727247,0.005544118,0.01524218,0.03252784,0.1978626,0.01153398,0.7117116],"study_design_scores_gemma":[0.0001412515,0.0008547334,0.04867603,0.0009242423,0.000597876,0.003455705,0.006150487,0.1931046,0.0275555,0.5649153,0.1532744,0.00034994],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1784052,0.008344221,0.7790045,0.000891666,0.001022242,0.0007867875,0.005516977,0.001510911,0.02451753],"genre_scores_gemma":[0.6261452,0.002929937,0.3538626,0.0002804649,0.0009608389,0.001110517,0.008528199,0.0003761158,0.005806125],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01387948,"threshold_uncertainty_score":0.01478523,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04907162359263313,"score_gpt":0.3432709032750194,"score_spread":0.2941992796823862,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}