{"id":"W4386576695","doi":"10.18653/v1/2023.mwe-1.1","title":"Token-level Identification of Multiword Expressions using Pre-trained Multilingual Language Models","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of New Brunswick","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Security token; Identification (biology); Natural language processing; Artificial intelligence; Linguistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001594571,0.001770062,0.0008754097,0.0008756933,0.0006331505,0.001929938,0.001430127,0.0008564395,0.003870505],"category_scores_gemma":[0.005353811,0.0006330479,0.0009349888,0.0006795815,0.0005109424,0.004111366,0.00170334,0.002066022,0.005032605],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009875022,"about_ca_system_score_gemma":0.00150257,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00687361,"about_ca_topic_score_gemma":0.01178928,"domain_scores_codex":[0.998656,0.0003313804,0.00008145492,0.0006582115,0.0001345536,0.0001383893],"domain_scores_gemma":[0.9979461,0.0009397405,0.0001130421,0.0004589465,0.0004481034,0.00009404474],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001091999,0.0008185201,0.02050954,0.0004065136,0.000596671,0.0009087753,0.0006178773,0.1651447,0.08638346,0.004494448,0.007418386,0.7116091],"study_design_scores_gemma":[0.00002766443,0.000218859,0.003179298,0.00003150522,0.0001417341,0.0002838914,0.00029881,0.9440843,0.04262407,0.004471061,0.004565663,0.00007321496],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.298693,0.001315506,0.6723921,0.0005774871,0.0003853749,0.0002950044,0.001971529,0.01429504,0.01007508],"genre_scores_gemma":[0.7966148,0.000333199,0.1896281,0.0003045932,0.00009169993,0.000176675,0.005501613,0.0007363902,0.006612896],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.00687361,"threshold_uncertainty_score":0.01366723,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06292493646583523,"score_gpt":0.3594634093157567,"score_spread":0.2965384728499215,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}