{"id":"W3032025286","doi":"","title":"Cifu: a Frequency Lexicon of Hong Kong Cantonese.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Linguistic Variation and Morphology","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Lexicon; Computer science; Lexical database; Word lists by frequency; Natural language processing; Lexical diversity; Artificial intelligence; Linguistics; Word (group theory); Phonology; Speech recognition; Vocabulary","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007203024,0.001029606,0.0004768316,0.005183459,0.001328657,0.001685167,0.0007218004,0.0002851599,0.01379566],"category_scores_gemma":[0.00275724,0.0002790742,0.0002257788,0.005367296,0.000497779,0.001517551,0.0009699404,0.0003844304,0.002156494],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002684825,"about_ca_system_score_gemma":0.004822945,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.2457548,"about_ca_topic_score_gemma":0.2203944,"domain_scores_codex":[0.9996526,0.00007433435,0.00009531876,0.00006451681,0.00007331763,0.00003993783],"domain_scores_gemma":[0.9983336,0.0003899707,0.0001152934,0.000173135,0.0008681649,0.0001197609],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001007124,0.0002076141,0.1042228,0.004230725,0.0003074145,0.002820788,0.03421256,0.003625031,0.03339489,0.03629917,0.2434427,0.5362292],"study_design_scores_gemma":[0.0001178484,0.0001879266,0.3539337,0.0006457585,0.0004067233,0.002419744,0.01381011,0.009488307,0.008754553,0.00346575,0.606543,0.0002265708],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"other","genre_scores_codex":[0.5587788,0.004032266,0.04017689,0.000728969,0.0003660244,0.001185789,0.2705348,0.005719889,0.1184766],"genre_scores_gemma":[0.8344222,0.001026194,0.02854934,0.0001155358,0.00004821115,0.0008829161,0.1152545,0.0006968945,0.01900418],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.2457548,"threshold_uncertainty_score":0.4886487,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0484546324202565,"score_gpt":0.3496756248661653,"score_spread":0.3012209924459088,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}