{"id":"W4389524278","doi":"10.18653/v1/2023.findings-emnlp.180","title":"CASE: Commonsense-Augmented Score with an Expanded Answer Space","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Commonsense knowledge; Space (punctuation); Commonsense reasoning; Weighting; Artificial intelligence; Question answering; Measure (data warehouse); Natural language processing; Machine learning; Data mining; Domain knowledge","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001935463,0.0001054854,0.0001033388,0.00009296751,0.000112812,0.0001026866,0.0003068774,0.00003387157,0.00002156259],"category_scores_gemma":[0.000008774859,0.00007975651,0.00001931524,0.0004845561,0.00002308514,0.0003974849,0.0001505004,0.00008329692,0.00007157887],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000190147,"about_ca_system_score_gemma":0.00003489137,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002539604,"about_ca_topic_score_gemma":0.0003634455,"domain_scores_codex":[0.999038,0.00004810037,0.0001082025,0.0002930298,0.000206954,0.0003056758],"domain_scores_gemma":[0.9990343,0.00004738581,0.00002747349,0.0007143189,0.00004048085,0.0001360091],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002151031,0.0009847851,0.09864166,0.0001959017,0.0003440114,0.05381028,0.04154976,0.01885626,0.006968835,0.5280977,0.07663134,0.1737043],"study_design_scores_gemma":[0.0008654354,0.0002539025,0.00188371,0.00003334287,0.000007733885,0.004069591,0.001246775,0.9854583,0.003466863,0.000678428,0.001599576,0.0004364066],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6256654,0.000007377944,0.3710744,0.0008655027,0.0001325918,0.00009610647,6.299184e-7,0.0006312656,0.00152671],"genre_scores_gemma":[0.9417551,0.000001676062,0.05567926,0.0003112121,0.00004617039,0.000008445058,0.000002816643,0.000009979795,0.002185313],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.966602,"threshold_uncertainty_score":0.3252376,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06141825289089449,"score_gpt":0.2746300107929852,"score_spread":0.2132117579020907,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}