{"id":"W1993536130","doi":"10.1145/564376.564438","title":"Using self-supervised word segmentation in Chinese information retrieval","year":2002,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Artificial intelligence; Character (mathematics); Word (group theory); Text segmentation; Natural language processing; Segmentation; Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007383575,0.0005472685,0.0007344923,0.002094724,0.0006787533,0.000624511,0.000814156,0.0005054069,0.001192782],"category_scores_gemma":[0.002138242,0.0003380737,0.0006815377,0.001944136,0.000824133,0.001960292,0.0006299192,0.0004238201,0.001107888],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005676956,"about_ca_system_score_gemma":0.001222871,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005186132,"about_ca_topic_score_gemma":0.007109409,"domain_scores_codex":[0.9990408,0.0002520547,0.0001005159,0.0002532542,0.0002761335,0.00007715615],"domain_scores_gemma":[0.9986517,0.0004170499,0.0001717485,0.0002541248,0.0004528964,0.00005251624],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002501014,0.000223589,0.002937537,0.0003672608,0.000099673,0.0001512361,0.0004856695,0.03140765,0.08386096,0.006181807,0.005384399,0.86865],"study_design_scores_gemma":[0.00006779557,0.0002625513,0.003663837,0.00001823574,0.0001051435,0.0002657429,0.0001256125,0.9068989,0.07383682,0.006667986,0.008014505,0.00007291113],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1044522,0.001078032,0.885659,0.0001980643,0.00008690423,0.000271478,0.000148174,0.004236649,0.003869526],"genre_scores_gemma":[0.5501761,0.0005948682,0.4418972,0.0001985629,0.0002189222,0.0003592638,0.001060775,0.000346046,0.005148163],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005186132,"threshold_uncertainty_score":0.0103119,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03339622337105059,"score_gpt":0.2628235046659737,"score_spread":0.2294272812949231,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}