{"id":"W4385570952","doi":"10.18653/v1/2023.findings-acl.179","title":"Grounding the Lexical Substitution Task in Entailment","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Textual entailment; Logical consequence; Substitution (logic); Computer science; Natural language processing; Sentence; Artificial intelligence; Context (archaeology); Task (project management); Relation (database); Word (group theory); Linguistics; Programming language; Data mining","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006218716,0.001406377,0.002178864,0.00337693,0.00229969,0.003108642,0.003001539,0.002183745,0.005581841],"category_scores_gemma":[0.03030574,0.0008922329,0.001655859,0.003980956,0.002394356,0.01112203,0.007047263,0.002367026,0.002622248],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001229201,"about_ca_system_score_gemma":0.002498887,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002906786,"about_ca_topic_score_gemma":0.006937806,"domain_scores_codex":[0.9888366,0.0052045,0.00129862,0.002118391,0.002172896,0.0003689056],"domain_scores_gemma":[0.9837614,0.008334717,0.0009254197,0.005171707,0.001612304,0.0001944701],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00173371,0.0007167404,0.01301086,0.002911341,0.0004742688,0.001411193,0.003103596,0.03032925,0.04718223,0.1784213,0.03098945,0.689716],"study_design_scores_gemma":[0.0002381072,0.0005397232,0.006126823,0.0004831495,0.0004249179,0.001987727,0.00212062,0.4403436,0.10371,0.3614845,0.08231746,0.0002232959],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1434995,0.002277758,0.8297484,0.001465734,0.000337437,0.000604736,0.00343817,0.005995969,0.01263229],"genre_scores_gemma":[0.4078676,0.000515623,0.5794479,0.000407771,0.0001401747,0.0004185548,0.008300397,0.0006357838,0.00226615],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006218716,"threshold_uncertainty_score":0.03288811,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02323847001305867,"score_gpt":0.2984554879231073,"score_spread":0.2752170179100487,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}