{"id":"W3177727271","doi":"10.1109/access.2021.3135807","title":"Comparative Analysis of Word Embeddings in Assessing Semantic Similarity of Complex Sentences","year":2021,"lang":"en","type":"article","venue":"IEEE Access","topic":"Topic Modeling","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"Lakehead University","funders":"Lakehead University","keywords":"Computer science; Natural language processing; Artificial intelligence; Semantic similarity; Word embedding; Sentence; Benchmark (surveying); Readability; Word (group theory); Similarity (geometry); Transformer; Embedding; Linguistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004816717,0.001202309,0.0007706962,0.004463241,0.0004189328,0.001486616,0.0007057505,0.0008573422,0.0009227282],"category_scores_gemma":[0.02600925,0.0001693649,0.0008079581,0.003197209,0.0005393604,0.003225791,0.001389438,0.000822308,0.0007220388],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004793477,"about_ca_system_score_gemma":0.0004826544,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001396739,"about_ca_topic_score_gemma":0.002819163,"domain_scores_codex":[0.9957841,0.002103635,0.0005077418,0.0007941853,0.0006595367,0.0001508578],"domain_scores_gemma":[0.9826909,0.01154502,0.001386087,0.001882999,0.002104066,0.0003909535],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003943911,0.001704217,0.1634812,0.003700729,0.00250628,0.0009163981,0.00266604,0.09962275,0.04524142,0.006964801,0.02835546,0.6408968],"study_design_scores_gemma":[0.0001602331,0.002953922,0.1053437,0.0002422604,0.0006810729,0.001577416,0.003087158,0.8294414,0.03137305,0.01361602,0.01130379,0.0002199697],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9234655,0.003785745,0.06000639,0.0003969219,0.0003112646,0.000187027,0.007245046,0.001703908,0.002898163],"genre_scores_gemma":[0.9379838,0.0008072317,0.03885845,0.00007564388,0.0001332785,0.0001376245,0.02118291,0.0001428009,0.000678306],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004816717,"threshold_uncertainty_score":0.02547354,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1432759075099785,"score_gpt":0.4075290179436385,"score_spread":0.26425311043366,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}