{"id":"W2963418282","doi":"10.1609/icwsm.v11i1.14859","title":"Data Sets: Word Embeddings Learned from Tweets and General Data","year":2017,"lang":"en","type":"article","venue":"Proceedings of the International AAAI Conference on Web and Social Media","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"ca_institutions":"Thomson Reuters (Canada)","funders":"","keywords":"Computer science; Word (group theory); Word embedding; Natural language processing; Embedding; Artificial intelligence; Representation (politics); Information retrieval; Sentiment analysis; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001509214,0.001488093,0.0006909809,0.001886738,0.0005465838,0.0008665386,0.001265566,0.001902814,0.003176558],"category_scores_gemma":[0.01313046,0.000453163,0.001540318,0.002569327,0.0008969465,0.002765133,0.001637176,0.002238253,0.002392533],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000763157,"about_ca_system_score_gemma":0.0006313791,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004067454,"about_ca_topic_score_gemma":0.005659868,"domain_scores_codex":[0.9984445,0.0004265759,0.0002071934,0.0004763845,0.000316431,0.0001288369],"domain_scores_gemma":[0.9946398,0.002116408,0.0004126718,0.00154529,0.001042621,0.0002432958],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.004989858,0.004544576,0.1445396,0.003847954,0.00172698,0.00295562,0.001343517,0.1252171,0.02357756,0.007783481,0.1601963,0.5192775],"study_design_scores_gemma":[0.0009424738,0.002261334,0.1340108,0.0003877761,0.0008245008,0.004041003,0.001831226,0.63649,0.04859147,0.02600495,0.1440649,0.0005495427],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"dataset","genre_scores_codex":[0.6759546,0.002594461,0.1077713,0.002082671,0.001197636,0.001772483,0.1953583,0.005601882,0.007666619],"genre_scores_gemma":[0.5581009,0.0008408218,0.1182668,0.0005723351,0.0002391611,0.001911607,0.3150999,0.0002864479,0.004681952],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.004067454,"threshold_uncertainty_score":0.01062667,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1245390394014516,"score_gpt":0.3492956645279873,"score_spread":0.2247566251265358,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}