{"id":"W2099779943","doi":"10.3115/1621969.1621986","title":"SemEval-2010 task 8","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":497,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"SemEval; Computer science; Task (project management); Testbed; Natural language processing; Artificial intelligence; Information extraction; Natural language; Semantics (computer science); Information retrieval; World Wide Web; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005198946,0.004077274,0.002694404,0.00360242,0.002634084,0.004150004,0.004288187,0.00579035,0.03894611],"category_scores_gemma":[0.01313454,0.0007238045,0.002497713,0.002474467,0.001281862,0.006731471,0.006502693,0.003865096,0.03491705],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002236745,"about_ca_system_score_gemma":0.004148105,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008651507,"about_ca_topic_score_gemma":0.0183302,"domain_scores_codex":[0.9941576,0.001845016,0.0005736018,0.001757894,0.001121457,0.0005444985],"domain_scores_gemma":[0.9934604,0.002662063,0.0003310729,0.001799342,0.001163988,0.0005829348],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0009239333,0.0007528941,0.002662512,0.002236014,0.0001892467,0.0007040164,0.0003737777,0.002035657,0.004587538,0.006513066,0.8471552,0.1318662],"study_design_scores_gemma":[0.00104636,0.0007268325,0.01202867,0.0007276273,0.0001957352,0.003927256,0.002426211,0.05422312,0.02605409,0.03052935,0.8678287,0.0002860241],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"other","genre_scores_codex":[0.1200891,0.007617997,0.07466604,0.008921039,0.007140889,0.004141478,0.5696771,0.07509587,0.1326504],"genre_scores_gemma":[0.09670084,0.0006602401,0.1012153,0.002068324,0.0004721157,0.002432297,0.763127,0.002673945,0.03064996],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.03894611,"threshold_uncertainty_score":0.1302878,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009039731999430325,"score_gpt":0.2638369322279143,"score_spread":0.254797200228484,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}