{"id":"W2911338114","doi":"10.1109/access.2019.2891692","title":"Challenging the Boundaries of Unsupervised Learning for Semantic Similarity","year":2019,"lang":"en","type":"article","venue":"IEEE Access","topic":"Topic Modeling","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"ca_institutions":"Lakehead University","funders":"","keywords":"Computer science; Semantic similarity; Artificial intelligence; Similarity (geometry); Natural language processing; Benchmark (surveying); Sentence; Word (group theory); Unsupervised learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01924324,0.00120061,0.002217489,0.00391062,0.001990312,0.004541766,0.003675702,0.003229158,0.001033353],"category_scores_gemma":[0.07397843,0.0006173057,0.001462181,0.003318342,0.00343293,0.009493399,0.005573011,0.004522084,0.0009004308],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001975898,"about_ca_system_score_gemma":0.002565922,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002719917,"about_ca_topic_score_gemma":0.0027231,"domain_scores_codex":[0.9803116,0.01100713,0.001144665,0.004478889,0.002682047,0.0003755496],"domain_scores_gemma":[0.9313058,0.0528606,0.002479249,0.007298245,0.005120513,0.0009356277],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004333513,0.000819092,0.01513732,0.0007408742,0.0006018046,0.0002327939,0.001266933,0.1614501,0.004190627,0.1623956,0.01395839,0.6387732],"study_design_scores_gemma":[0.000025107,0.00008528871,0.001694393,0.00005450784,0.00002372207,0.0001111898,0.0002378236,0.7916578,0.001630704,0.2022901,0.002159198,0.00003022777],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0330237,0.0009364685,0.9614868,0.001214091,0.00007834824,0.00016739,0.0002117185,0.0005444471,0.002337076],"genre_scores_gemma":[0.6270418,0.0007656063,0.3669375,0.0005741809,0.000530622,0.0005360475,0.001518247,0.0004029766,0.001692917],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01924324,"threshold_uncertainty_score":0.1017692,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04456077407634325,"score_gpt":0.2982622546993254,"score_spread":0.2537014806229822,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}