{"id":"W42051811","doi":"","title":"Waterloo at NTCIR-3: Using Self-supervised Word Segmentation","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Artificial intelligence; Text segmentation; Natural language processing; Segmentation; Word (group theory); Task (project management); Character (mathematics); Machine translation; Information retrieval","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003574012,0.001626136,0.001445356,0.003389961,0.001651161,0.002029569,0.001412548,0.001169305,0.01172363],"category_scores_gemma":[0.006012942,0.000665681,0.0006870902,0.001751547,0.0008244959,0.003603101,0.001842163,0.000823083,0.0102799],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001193986,"about_ca_system_score_gemma":0.002636195,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.04488151,"about_ca_topic_score_gemma":0.05513894,"domain_scores_codex":[0.99619,0.001668792,0.0002757554,0.0007549904,0.0008585224,0.0002520489],"domain_scores_gemma":[0.9964805,0.0009977145,0.0001622945,0.0006550174,0.001453965,0.0002504766],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001350952,0.000694398,0.005754907,0.001854611,0.0003308822,0.0007393436,0.001605714,0.006615328,0.1222996,0.004558468,0.3025504,0.5516453],"study_design_scores_gemma":[0.001340656,0.001146727,0.01968316,0.0002503001,0.0004678258,0.001395499,0.002256315,0.3080784,0.2934982,0.01102316,0.3602673,0.0005924586],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1336198,0.003384731,0.5061527,0.001972443,0.0016259,0.003939837,0.04447125,0.2349445,0.06988887],"genre_scores_gemma":[0.2613272,0.000745352,0.5640775,0.0008956865,0.0002987292,0.002063254,0.1264137,0.00633345,0.03784518],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.04488151,"threshold_uncertainty_score":0.08924055,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02440267527869397,"score_gpt":0.2597840563686933,"score_spread":0.2353813810899993,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}