{"id":"W2991485158","doi":"10.48550/arxiv.1911.13280","title":"Deconstructing and reconstructing word embedding algorithms","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Word2vec; Pointwise mutual information; Word (group theory); Word embedding; Computer science; Pointwise; Embedding; Algorithm; Feature (linguistics); Construct (python library); Artificial intelligence; Natural language processing; Mutual information; Mathematics; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00321891,0.001389967,0.0009375191,0.001227089,0.0005084475,0.001785045,0.001333569,0.001561002,0.001476413],"category_scores_gemma":[0.01802256,0.0005300082,0.0008176398,0.001075114,0.001532525,0.005234823,0.003218341,0.002789199,0.00172387],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006321766,"about_ca_system_score_gemma":0.001070906,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009767774,"about_ca_topic_score_gemma":0.00184654,"domain_scores_codex":[0.9977064,0.001031163,0.0001917832,0.0005683743,0.000370564,0.0001317087],"domain_scores_gemma":[0.9947848,0.0024134,0.0003115476,0.001493012,0.0008628471,0.0001344055],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003228331,0.0002197398,0.004679695,0.0003186225,0.0001264528,0.0001320715,0.0006703159,0.2804942,0.02206226,0.08380237,0.005277051,0.6018943],"study_design_scores_gemma":[0.00002832819,0.0001050576,0.00042195,0.00002601188,0.00001325825,0.00009905877,0.0001187748,0.9211272,0.01368704,0.06114552,0.003200886,0.0000267948],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03179516,0.0003209357,0.9656257,0.0002952729,0.00004607139,0.00004233175,0.0001262152,0.001005856,0.0007425129],"genre_scores_gemma":[0.3537287,0.0004844667,0.639541,0.0002339524,0.0001139548,0.0002364991,0.001599975,0.000507512,0.003554127],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.00321891,"threshold_uncertainty_score":0.01702338,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07496298646145166,"score_gpt":0.2028099849556326,"score_spread":0.1278469984941809,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}