{"id":"W3189820663","doi":"10.18653/v1/2022.naacl-main.178","title":"WiC = TSV = WSD: On the Equivalence of Three Semantic Tasks","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Computer science; SemEval; Equivalence (formal languages); Natural language processing; Pairwise comparison; Task (project management); Semantic equivalence; Artificial intelligence; Word (group theory); Context (archaeology); Popularity; Semantic similarity; Word-sense disambiguation; WordNet; Linguistics; Semantic Web; Semantic computing; Psychology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01615383,0.001340007,0.002119208,0.00314033,0.002929813,0.008039965,0.00335785,0.004121046,0.007306277],"category_scores_gemma":[0.06776414,0.00065767,0.002549388,0.003253804,0.006065552,0.02079933,0.01347554,0.007106788,0.002335951],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002316904,"about_ca_system_score_gemma":0.005624144,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006208922,"about_ca_topic_score_gemma":0.003977051,"domain_scores_codex":[0.9816098,0.007332444,0.001766384,0.005154795,0.003113075,0.001023441],"domain_scores_gemma":[0.9551606,0.02497901,0.002510502,0.01184621,0.004111151,0.001392569],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001130761,0.0004670591,0.008509446,0.000808361,0.0001988593,0.0002139704,0.001497333,0.01489149,0.004202111,0.5940169,0.02474216,0.3493215],"study_design_scores_gemma":[0.0001041398,0.000213902,0.001979668,0.00009575088,0.0000706722,0.0002561315,0.0006712707,0.1382637,0.003913534,0.8439744,0.01039031,0.00006651175],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06741209,0.001117026,0.8991251,0.006334496,0.0005945058,0.0007198582,0.00208714,0.002793603,0.01981627],"genre_scores_gemma":[0.4523533,0.0007585635,0.5268336,0.001607852,0.0004760148,0.001123689,0.00956057,0.0009860203,0.006300421],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01615383,"threshold_uncertainty_score":0.08543062,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.021952110168749,"score_gpt":0.2601831011040238,"score_spread":0.2382309909352748,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}