{"id":"W3117912976","doi":"10.18653/v1/2020.semeval-1.32","title":"UAlberta at SemEval-2020 Task 2: Using Translations to Predict Cross-Lingual Entailment","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Leverage (statistics); Computer science; Textual entailment; Logical consequence; Natural language processing; SemEval; Artificial intelligence; Task (project management); Word (group theory); Linguistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001269895,0.0001727225,0.000157124,0.00005524514,0.0002150809,0.0002862755,0.0008885031,0.00007061115,0.0001254673],"category_scores_gemma":[0.00009332634,0.0001515847,0.00007241783,0.0005839237,0.00003153759,0.0004759069,0.0004826016,0.0001435838,0.00006944002],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001014772,"about_ca_system_score_gemma":0.00008782367,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007302136,"about_ca_topic_score_gemma":0.00003078102,"domain_scores_codex":[0.9984555,0.0000350851,0.0002842486,0.0005383199,0.0003837329,0.0003031016],"domain_scores_gemma":[0.9991894,0.00005575721,0.00005792348,0.0003425015,0.00008823663,0.0002661607],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001654702,0.0002741456,0.01220055,0.0002728116,0.000163246,0.0002725261,0.04804839,0.003876886,0.7934073,0.03100357,0.02127667,0.08903846],"study_design_scores_gemma":[0.001040892,0.0005150128,0.0006664063,0.0001299078,0.00004989476,0.00009451741,0.00008135434,0.4103144,0.5560305,0.00488661,0.0249255,0.001265092],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08710987,0.0004103107,0.9019131,0.00809598,0.0001524991,0.0003653971,0.00001150623,0.001112867,0.0008284597],"genre_scores_gemma":[0.5643813,0.000001592482,0.4322247,0.002972245,0.0001004063,0.00000790223,0.000004720041,0.00001181193,0.0002953498],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.4772714,"threshold_uncertainty_score":0.6181445,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02719183343376473,"score_gpt":0.3227681545403572,"score_spread":0.2955763211065924,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}