{"id":"W3099354896","doi":"10.18653/v1/2020.emnlp-main.681","title":"Deconstructing word embedding algorithms","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Word2vec; Word (group theory); Computer science; Word embedding; Natural language processing; Embedding; Feature (linguistics); Artificial intelligence; Resource (disambiguation); Quality (philosophy); Linguistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003543407,0.001536628,0.0009748297,0.001622721,0.0005141366,0.002395439,0.001351277,0.001355551,0.002280987],"category_scores_gemma":[0.01827777,0.0006764589,0.001118788,0.001428024,0.001645114,0.006630545,0.003500678,0.003462956,0.001789448],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007102387,"about_ca_system_score_gemma":0.0008506796,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001495766,"about_ca_topic_score_gemma":0.002568572,"domain_scores_codex":[0.9972385,0.001325828,0.0002146566,0.0006453217,0.0004214706,0.000154206],"domain_scores_gemma":[0.9936231,0.003333776,0.000311544,0.001584268,0.001006432,0.0001409403],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001807759,0.0001192183,0.00315689,0.0003226365,0.0001513984,0.0001132391,0.0007595668,0.2522245,0.007777504,0.2662317,0.005494153,0.4634685],"study_design_scores_gemma":[0.00001331263,0.00005991125,0.000339388,0.00005642872,0.00001823837,0.00007283119,0.0001017072,0.7615878,0.003669934,0.2282725,0.005784796,0.00002319172],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01094121,0.0004898304,0.9865751,0.0003371504,0.00004766717,0.00004310721,0.0001181203,0.0004572567,0.0009906701],"genre_scores_gemma":[0.3178557,0.002101167,0.6683693,0.0004243406,0.00026949,0.0004053371,0.002260236,0.0007736381,0.007540901],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003543407,"threshold_uncertainty_score":0.01873952,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05486440649495251,"score_gpt":0.3044185296808561,"score_spread":0.2495541231859036,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}