{"id":"W4389524352","doi":"10.18653/v1/2023.findings-emnlp.861","title":"COMET-M: Reasoning about Multiple Events in Complex Sentences","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Comet; Event (particle physics); Computer science; Natural language processing; Coreference; Sentence; Context (archaeology); Inference; Artificial intelligence; Resolution (logic); Meaning (existential); Natural (archaeology); Commonsense knowledge; Natural language; Psychology; Knowledge-based systems; History; Physics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000344744,0.00007699038,0.0001151253,0.0001641754,0.00006177806,0.00003573182,0.0006181182,0.00002776973,0.00003070524],"category_scores_gemma":[0.00008338994,0.00006883059,0.00003012916,0.0006971597,0.0000116903,0.0002456298,0.0003212026,0.00007999561,0.0002110904],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002518161,"about_ca_system_score_gemma":0.00002044352,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000517457,"about_ca_topic_score_gemma":0.0003228862,"domain_scores_codex":[0.9989703,0.00004567174,0.0001897828,0.0002906946,0.0002242367,0.0002792694],"domain_scores_gemma":[0.9994556,0.000110161,0.00003315408,0.0003266373,0.00002310098,0.00005135737],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000003325651,0.00005808548,0.9045722,0.00002360734,0.00001274066,0.00004940281,0.001966784,0.009807991,0.0009367185,0.03899185,0.003175621,0.04040166],"study_design_scores_gemma":[0.000188821,0.000005722799,0.2185752,0.00001888521,3.252478e-7,0.000002246951,0.0001007631,0.7790455,0.00005900088,0.001233023,0.0006946604,0.00007592324],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5412637,0.00002306889,0.4514173,0.001425197,0.0004185288,0.0001435869,7.555036e-7,0.0005487252,0.004759122],"genre_scores_gemma":[0.942215,0.00000791582,0.05653837,0.0002165827,0.00003271029,0.000008751317,0.000003186151,0.000004104722,0.0009733959],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7692375,"threshold_uncertainty_score":0.280683,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06975010678736791,"score_gpt":0.2996717027449558,"score_spread":0.2299215959575879,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}