{"id":"W3183679143","doi":"10.18653/v1/2021.mwe-1.1","title":"A Long Hard Look at MWEs in the Age of Language Models","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Meaning (existential); Context (archaeology); Representation (politics); Layer (electronics); Language model; History; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007431066,0.001640325,0.00204529,0.00268185,0.001661433,0.008547797,0.002918804,0.004221783,0.01135711],"category_scores_gemma":[0.04454133,0.001323088,0.001272838,0.002433561,0.007492975,0.04398726,0.005323106,0.01657364,0.008717666],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001938801,"about_ca_system_score_gemma":0.001338728,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002695845,"about_ca_topic_score_gemma":0.002713007,"domain_scores_codex":[0.9957059,0.001736706,0.0002368061,0.001020445,0.001152648,0.0001474443],"domain_scores_gemma":[0.977419,0.01568348,0.0006659451,0.003077122,0.002472252,0.0006822309],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002255248,0.0001046278,0.002256351,0.0007846219,0.0002659756,0.0002221967,0.0008588904,0.01069072,0.001276315,0.6774487,0.07946989,0.2263962],"study_design_scores_gemma":[0.00001468583,0.00004623542,0.0003738106,0.000335175,0.00003057313,0.0002203296,0.0003575014,0.02532518,0.0006599732,0.8401357,0.1324287,0.00007206154],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008837139,0.07384731,0.632966,0.2466631,0.00631293,0.00006191245,0.001505489,0.002764002,0.02704209],"genre_scores_gemma":[0.272781,0.1115956,0.4418255,0.07226671,0.03167444,0.0003789906,0.004120875,0.005460745,0.05989622],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01135711,"threshold_uncertainty_score":0.03929967,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02002878756132303,"score_gpt":0.273988851447257,"score_spread":0.253960063885934,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}