{"id":"W2886752202","doi":"10.1139/geomat-2018-0007","title":"A cyclic self-learning Chinese word segmentation for the geoscience domain","year":2018,"lang":"en","type":"article","venue":"GEOMATICA","topic":"Topic Modeling","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"National Key Research and Development Program of China; National Natural Science Foundation of China","keywords":"Computer science; Context (archaeology); Domain (mathematical analysis); Benchmark (surveying); Natural language processing; Text segmentation; Segmentation; Word (group theory); Artificial intelligence; Information retrieval; Geography; Archaeology; Linguistics; Cartography","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006259039,0.001242863,0.001191711,0.002609696,0.001034861,0.0008351715,0.001536254,0.001131064,0.005080044],"category_scores_gemma":[0.001599911,0.0004730142,0.001223634,0.002795712,0.0005041297,0.002041991,0.001454591,0.001218514,0.002518465],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007321265,"about_ca_system_score_gemma":0.002279312,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01721934,"about_ca_topic_score_gemma":0.02416062,"domain_scores_codex":[0.9994013,0.00009290831,0.000048349,0.0002813433,0.00009562544,0.00008032022],"domain_scores_gemma":[0.999041,0.0003933807,0.00004473581,0.0001463988,0.0002850767,0.00008925008],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007856818,0.0003332273,0.002585096,0.0003599855,0.000138696,0.0002976171,0.0003410847,0.03100223,0.03022595,0.006131897,0.01997027,0.9078282],"study_design_scores_gemma":[0.00007626951,0.0001617904,0.001461969,0.00002236871,0.00009029553,0.0001474399,0.0001663574,0.9715638,0.01171242,0.007403111,0.007159408,0.00003485302],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09494776,0.001893042,0.8854444,0.0004458952,0.0003376952,0.0002369759,0.002009002,0.01052493,0.004160366],"genre_scores_gemma":[0.3667708,0.0009005271,0.6066256,0.0003435914,0.0003354036,0.0003293261,0.01235041,0.001153708,0.01119069],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01721934,"threshold_uncertainty_score":0.03423822,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0110331818280587,"score_gpt":0.2655054943658863,"score_spread":0.2544723125378276,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}