{"id":"W4404787914","doi":"10.1109/access.2024.3507382","title":"Enhancing Sindhi Word Segmentation Using Subword Representation Learning and Position-Aware Self-Attention","year":2024,"lang":"en","type":"article","venue":"IEEE Access","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"National Key Research and Development Program of China","keywords":"Computer science; Natural language processing; Representation (politics); Segmentation; Artificial intelligence; Speech recognition; Word (group theory); Text segmentation; Position (finance); Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006646829,0.001755803,0.001269416,0.002238732,0.0006166644,0.00112523,0.001340702,0.001293386,0.003145849],"category_scores_gemma":[0.001789832,0.0003836906,0.001016535,0.002212877,0.0004558336,0.003211017,0.001240585,0.001443593,0.004377891],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007332014,"about_ca_system_score_gemma":0.001150708,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006589488,"about_ca_topic_score_gemma":0.0142873,"domain_scores_codex":[0.9994408,0.00009148652,0.00004903982,0.0002654572,0.0000837058,0.00006937231],"domain_scores_gemma":[0.9991899,0.0002837921,0.0000851801,0.0001895171,0.0002072511,0.00004441275],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004886135,0.0003372323,0.00289583,0.0004962319,0.0001498741,0.0002253351,0.0004237745,0.02004954,0.0556511,0.002656142,0.02211339,0.8945128],"study_design_scores_gemma":[0.00006494264,0.0002928628,0.004132352,0.00004316391,0.0001486939,0.0003126064,0.0003386526,0.9216059,0.05380746,0.006693912,0.01250083,0.00005865383],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3088849,0.006189094,0.6229783,0.0007904312,0.0007294301,0.0004477199,0.003953099,0.04198951,0.01403745],"genre_scores_gemma":[0.6021292,0.001279848,0.3557492,0.0006111951,0.0002826805,0.0004020325,0.01862619,0.001259016,0.0196607],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006589488,"threshold_uncertainty_score":0.01310229,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03326843506076371,"score_gpt":0.3456948562904092,"score_spread":0.3124264212296455,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}