{"id":"W2991550749","doi":"","title":"SemEval-2010 Task 8: Multi-Way Classification of Semantic Relations Between Pairs of Nominals","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"SemEval; Task (project management); Computer science; Testbed; Natural language processing; Artificial intelligence; Information retrieval; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01034599,0.006004736,0.003795773,0.004478743,0.003058824,0.004620779,0.006107028,0.007886427,0.02330754],"category_scores_gemma":[0.02230769,0.001173424,0.00353648,0.002985087,0.00202169,0.007457794,0.008370015,0.005820986,0.02347517],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002429326,"about_ca_system_score_gemma":0.003833299,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007068848,"about_ca_topic_score_gemma":0.01128763,"domain_scores_codex":[0.9889642,0.003977952,0.000751466,0.003815502,0.001610996,0.0008798306],"domain_scores_gemma":[0.9839866,0.007935614,0.0008251827,0.003787406,0.002108906,0.001356261],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.004390146,0.00353908,0.01100422,0.003661097,0.0008327474,0.001136309,0.001253988,0.005256352,0.01696527,0.004110279,0.6164522,0.3313983],"study_design_scores_gemma":[0.005263818,0.00551417,0.0632395,0.001260048,0.0008439278,0.007977816,0.006918337,0.225517,0.1052424,0.03963151,0.5377135,0.0008779311],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4036624,0.008332317,0.1217123,0.005348884,0.007401189,0.008474827,0.2360008,0.1394914,0.0695759],"genre_scores_gemma":[0.2729555,0.000742573,0.227097,0.002137124,0.000673885,0.00399131,0.4568724,0.004514673,0.03101555],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02330754,"threshold_uncertainty_score":0.07797146,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09263992999547042,"score_gpt":0.2316887088823288,"score_spread":0.1390487788868583,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}