{"id":"W4389524352","doi":"10.18653/v1/2023.findings-emnlp.861","title":"COMET-M: Reasoning about Multiple Events in Complex Sentences","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Comet; Event (particle physics); Computer science; Natural language processing; Coreference; Sentence; Context (archaeology); Inference; Artificial intelligence; Resolution (logic); Meaning (existential); Natural (archaeology); Commonsense knowledge; Natural language; Psychology; Knowledge-based systems; History; Physics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002544888,0.002492422,0.0008163151,0.002784206,0.001317665,0.002939324,0.00452328,0.002874738,0.0105085],"category_scores_gemma":[0.01949675,0.0009636763,0.003746504,0.001229013,0.001044599,0.007596805,0.003861411,0.004191986,0.00304707],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001555523,"about_ca_system_score_gemma":0.001689773,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007242864,"about_ca_topic_score_gemma":0.0189602,"domain_scores_codex":[0.997407,0.0005949053,0.0001895224,0.001188277,0.0005113934,0.0001089062],"domain_scores_gemma":[0.9914471,0.006309489,0.0004069598,0.0009835124,0.000594011,0.0002589388],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001129653,0.0007506257,0.01540913,0.004935651,0.001436236,0.003406271,0.003524753,0.08848444,0.02548243,0.06628019,0.1744198,0.6147408],"study_design_scores_gemma":[0.0002068403,0.0001781407,0.003552796,0.0002918906,0.0003559389,0.001319015,0.0006703142,0.8113651,0.01877027,0.1059327,0.05723168,0.0001253138],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03227485,0.001745874,0.8829681,0.002140202,0.0004442436,0.0009134753,0.02417469,0.04629061,0.009048043],"genre_scores_gemma":[0.2239479,0.0006412828,0.7163821,0.001402733,0.0003009346,0.0006100276,0.05152361,0.00149988,0.00369148],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0105085,"threshold_uncertainty_score":0.03515446,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06975010678736791,"score_gpt":0.2996717027449558,"score_spread":0.2299215959575879,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}