{"id":"W2991550749","doi":"","title":"SemEval-2010 Task 8: Multi-Way Classification of Semantic Relations Between Pairs of Nominals","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"SemEval; Task (project management); Computer science; Testbed; Natural language processing; Artificial intelligence; Information retrieval; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004720782,0.000250339,0.0004743692,0.0004809503,0.00006598034,0.00003549725,0.001922279,0.0004227759,0.00001169048],"category_scores_gemma":[0.0001211813,0.0002736177,0.0002171303,0.0006635609,0.0001519771,0.000405088,0.001234901,0.0005666878,0.00003007795],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001431882,"about_ca_system_score_gemma":0.0002255022,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001923901,"about_ca_topic_score_gemma":0.0000197615,"domain_scores_codex":[0.9982726,0.0001755229,0.0004122273,0.0007813119,0.0001425275,0.000215788],"domain_scores_gemma":[0.9967927,0.0002418463,0.0009850968,0.001480774,0.0004256477,0.00007390806],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009147038,0.0009528291,0.4647145,0.003845149,0.0008965142,0.0001092752,0.00392053,0.03783685,0.04667813,0.4311916,0.001747757,0.008015391],"study_design_scores_gemma":[0.00096686,0.0001739579,0.1070649,0.001439741,0.0004848924,0.00000364624,0.0001605074,0.7562281,0.02801848,0.1041973,0.00009284766,0.001168739],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.175732,0.0002271914,0.8229373,0.00009436439,0.0001878459,0.0003588037,0.00003787457,0.0002198874,0.0002047026],"genre_scores_gemma":[0.911949,0.00003028388,0.08679011,0.000008186829,0.00002278969,8.815632e-7,0.00003139331,0.00001581214,0.001151588],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7362169,"threshold_uncertainty_score":0.9999716,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09263992999547042,"score_gpt":0.2316887088823288,"score_spread":0.1390487788868583,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}