{"id":"W4413060628","doi":"10.1145/3748316","title":"Inner-character and Inner-word Features Based Representation Learning for Chinese Word Embedding","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Asian and Low-Resource Language Information Processing","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Fundamental Research Funds for the Central Universities; National Key Research and Development Program of China; Natural Science Foundation of Sichuan Province; China Postdoctoral Science Foundation; National Natural Science Foundation of China","keywords":"Pinyin; Computer science; Artificial intelligence; Natural language processing; Word (group theory); Character (mathematics); Word embedding; Feature (linguistics); Similarity (geometry); Chinese characters; Speech recognition; Embedding; Linguistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004026518,0.001188458,0.0007506169,0.0009139029,0.0003005513,0.0005149595,0.0008374139,0.0004832695,0.001837907],"category_scores_gemma":[0.001544365,0.0002639682,0.00077053,0.001630798,0.0004057441,0.002382272,0.0009714679,0.001148956,0.000909448],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000415084,"about_ca_system_score_gemma":0.0008894226,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003789002,"about_ca_topic_score_gemma":0.006093041,"domain_scores_codex":[0.9995568,0.00009982144,0.00003905285,0.0001703793,0.0000793573,0.00005444247],"domain_scores_gemma":[0.9995205,0.0001432897,0.00004718708,0.0001035835,0.0001532299,0.0000321297],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001711837,0.0002080612,0.003434965,0.0002126105,0.0001206477,0.0001082341,0.0002006688,0.07439797,0.01416263,0.01006712,0.01049501,0.8864209],"study_design_scores_gemma":[0.00001434393,0.00009720097,0.0008987217,0.00001111381,0.00003426331,0.0000579027,0.00005044096,0.9853209,0.00389843,0.007786679,0.001810044,0.00001999764],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09279989,0.001268327,0.9007471,0.0003122231,0.000156171,0.00008906791,0.0006776812,0.001965865,0.001983771],"genre_scores_gemma":[0.8096068,0.001291546,0.1757347,0.0002273718,0.000188819,0.0002579743,0.00477657,0.0001984936,0.007717684],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003789002,"threshold_uncertainty_score":0.007533908,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.006695338832000711,"score_gpt":0.2708147697501971,"score_spread":0.2641194309181965,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}