{"id":"W2077818552","doi":"10.1162/coli.2010.36.1.36104","title":"Automatically Identifying the Source Words of Lexical Blends in English","year":2010,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canada Research Chairs; University of Toronto","funders":"University of Toronto","keywords":"Computer science; Natural language processing; Lexicon; Artificial intelligence; Task (project management); Set (abstract data type); Identification (biology); Word (group theory); Source text; Linguistics; Programming language","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001054521,0.000705841,0.0005604802,0.0021539,0.0008432464,0.001372754,0.0005703391,0.0006234822,0.004629331],"category_scores_gemma":[0.004873028,0.0005177396,0.0004955903,0.001638864,0.0007904175,0.004932688,0.001313445,0.0006782518,0.001779746],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005172645,"about_ca_system_score_gemma":0.0004126126,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003005474,"about_ca_topic_score_gemma":0.004736926,"domain_scores_codex":[0.9990075,0.0002442419,0.0001882162,0.000358569,0.0001485434,0.00005304737],"domain_scores_gemma":[0.9957615,0.002716227,0.0005959516,0.0003164616,0.0005270848,0.00008294561],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001730852,0.0003958741,0.1301558,0.003000718,0.0002253904,0.004543259,0.01285499,0.006506252,0.1806267,0.01509992,0.01548809,0.6293721],"study_design_scores_gemma":[0.0002342961,0.0006191266,0.3282594,0.0006218338,0.0005810839,0.01104208,0.01911733,0.3299426,0.179192,0.03495542,0.095079,0.0003557636],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.9248342,0.001066625,0.06256165,0.0004195592,0.00008583207,0.00011024,0.003035278,0.001577151,0.00630935],"genre_scores_gemma":[0.9448154,0.0004835913,0.0476118,0.00007942822,0.00004621991,0.00006215484,0.004107028,0.0003303954,0.002463989],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.004629331,"threshold_uncertainty_score":0.01548672,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01327449925434027,"score_gpt":0.2962019672436924,"score_spread":0.2829274679893521,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}