{"id":"W2123215530","doi":"10.1162/coli.08-010-r1-07-048","title":"Unsupervised Type and Token Identification of Idiomatic Expressions","year":2009,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":201,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canada Research Chairs; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Atomic Energy of Canada Limited; University of Toronto","keywords":"Linguistics; Identification (biology); Interpretation (philosophy); Computer science; Natural language processing; Literal (mathematical logic); Task (project management); Context (archaeology); Security token; Artificial intelligence; Semantic property; Expression (computer science); Psychology; History","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001952591,0.000612154,0.0009856471,0.003605598,0.0006721235,0.00164894,0.001460423,0.0008356387,0.001613579],"category_scores_gemma":[0.01012901,0.0003166714,0.0007107049,0.002530779,0.0008239504,0.002931914,0.001063656,0.001126331,0.001534898],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006121513,"about_ca_system_score_gemma":0.001109624,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001350142,"about_ca_topic_score_gemma":0.00213235,"domain_scores_codex":[0.9975382,0.0006523495,0.0002863831,0.000813653,0.0004878339,0.0002216994],"domain_scores_gemma":[0.9904217,0.004739991,0.001297696,0.001365272,0.001884192,0.0002911735],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001272401,0.0002930535,0.08236528,0.0005782445,0.0001484841,0.0006804428,0.001315691,0.00578367,0.08303012,0.01692225,0.01054521,0.7970653],"study_design_scores_gemma":[0.00008678516,0.0002672633,0.07606187,0.00009126233,0.0002044139,0.002858156,0.001630861,0.7434557,0.112568,0.04964045,0.01293075,0.0002044469],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.402551,0.0005095452,0.5849693,0.000363638,0.0001920137,0.0003078166,0.002072431,0.004041636,0.00499272],"genre_scores_gemma":[0.63266,0.0002006652,0.3592974,0.0001009383,0.00009571162,0.0002496105,0.003979844,0.0004924405,0.002923427],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003605598,"threshold_uncertainty_score":0.01032639,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01564023181424613,"score_gpt":0.2969218139515016,"score_spread":0.2812815821372555,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}