{"id":"W2912277365","doi":"10.1109/bigdata.2018.8622463","title":"Deep Neural Networks for Social Media Word Segmentation of Asian Languages","year":2018,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Word (group theory); Artificial intelligence; Natural language processing; Social media; Artificial neural network; Character (mathematics); Segmentation; Task (project management); Text segmentation; Language model; Layer (electronics); Linguistics; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003977425,0.001225088,0.0004256301,0.001322733,0.0004931488,0.0006354231,0.0006078317,0.0006539109,0.002824005],"category_scores_gemma":[0.001256257,0.0003335919,0.0006272966,0.00160055,0.0003025987,0.001824388,0.0007181837,0.001171297,0.001638206],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007832794,"about_ca_system_score_gemma":0.0007442016,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01286608,"about_ca_topic_score_gemma":0.01785206,"domain_scores_codex":[0.9997466,0.00005831347,0.00002191829,0.00008584443,0.00003097608,0.00005632714],"domain_scores_gemma":[0.9996822,0.0001447516,0.00005162147,0.00002916989,0.0000730467,0.00001928378],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006790262,0.000354538,0.006776321,0.0002810458,0.0002008664,0.0004306622,0.0006440185,0.1744954,0.02446978,0.006961757,0.01562587,0.7690807],"study_design_scores_gemma":[0.000009943141,0.00002894338,0.001169742,0.00001353404,0.00002425408,0.00002479687,0.00008759177,0.9871982,0.004459496,0.004979046,0.001993332,0.00001111176],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4427762,0.004131419,0.5183073,0.00162941,0.0004396252,0.0002470273,0.004339057,0.01469109,0.0134387],"genre_scores_gemma":[0.88868,0.0008600777,0.09263729,0.0002593031,0.0001465062,0.0001872636,0.005935261,0.0002494076,0.01104491],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01286608,"threshold_uncertainty_score":0.02558237,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02608668151191412,"score_gpt":0.2924660520757331,"score_spread":0.2663793705638189,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}