{"id":"W2759162401","doi":"","title":"Named Entity Recognition and Hashtag Decomposition to Improve the Classification of Tweets","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université du Québec; Université du Québec à Montréal","funders":"","keywords":"WordNet; Computer science; Artificial intelligence; Natural language processing; Preprocessor; Named-entity recognition; Segmentation; Field (mathematics); Task (project management); Information retrieval; Semantics (computer science)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001644848,0.001080936,0.0007878874,0.005513307,0.0007404467,0.001717609,0.0008847698,0.001057576,0.002555892],"category_scores_gemma":[0.004966146,0.0002776225,0.001068448,0.004363397,0.0003321591,0.004296219,0.0009368116,0.001062694,0.004900513],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005720077,"about_ca_system_score_gemma":0.0007847367,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003634545,"about_ca_topic_score_gemma":0.005031471,"domain_scores_codex":[0.9986726,0.0004027334,0.0001734953,0.0003228578,0.0002769213,0.0001515159],"domain_scores_gemma":[0.996586,0.001370486,0.0003719011,0.0005233038,0.001017746,0.0001306011],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001135274,0.0007552558,0.02290549,0.0005453112,0.0003328935,0.0004969871,0.0008355009,0.01281829,0.07755624,0.00754678,0.02439554,0.8506764],"study_design_scores_gemma":[0.00007785489,0.0003662212,0.03187007,0.00009507802,0.0003697956,0.0006550906,0.001066421,0.7912063,0.1152668,0.01787667,0.04099265,0.0001571258],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2415081,0.002172107,0.7153081,0.001322585,0.0007874202,0.0006183474,0.005462853,0.02413374,0.008686766],"genre_scores_gemma":[0.5241119,0.0007530591,0.4504972,0.0003234281,0.0003081037,0.0002749267,0.01544021,0.0004746833,0.007816541],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005513307,"threshold_uncertainty_score":0.008698881,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0829395137147931,"score_gpt":0.3343411440966311,"score_spread":0.251401630381838,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}