{"id":"W4221053465","doi":"10.1162/tacl_a_00458","title":"Neuro-symbolic Natural Logic with Introspective Revision for Natural Language Inference","year":2022,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Interpretability; Computer science; Artificial intelligence; Inference; Generalization; Introspection; Spurious relationship; Machine learning; Rule of inference; Natural language; Natural (archaeology); Overfitting; Artificial neural network; Cognitive psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001843182,0.0005528256,0.000633723,0.0008508851,0.0004576654,0.001403981,0.001792758,0.0006745504,0.003133187],"category_scores_gemma":[0.006642176,0.0003335029,0.001132038,0.0005914749,0.001791209,0.002718461,0.001574409,0.002096638,0.0004030195],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001661842,"about_ca_system_score_gemma":0.002015925,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005793418,"about_ca_topic_score_gemma":0.008246122,"domain_scores_codex":[0.9987454,0.0005189743,0.00007245454,0.0002496143,0.000320226,0.00009310909],"domain_scores_gemma":[0.9969094,0.001712593,0.0002893679,0.0006043874,0.0003711364,0.0001130994],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002057715,0.0001885293,0.001793911,0.0002010745,0.0001403731,0.0003122473,0.0003263401,0.6272089,0.004886189,0.2020287,0.002742216,0.1599657],"study_design_scores_gemma":[0.00001037496,0.00001530027,0.00009419651,0.000008019105,0.000009638651,0.0000228229,0.000007046537,0.9390288,0.0006276523,0.05962204,0.0005472131,0.000006912875],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02058485,0.0002788542,0.9748824,0.0004582441,0.00003620764,0.0000611287,0.0001401086,0.001287956,0.0022702],"genre_scores_gemma":[0.7641208,0.0002218944,0.2333304,0.0002086576,0.00005914608,0.000157332,0.0002678611,0.0001130113,0.001520731],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005793418,"threshold_uncertainty_score":0.01205754,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01236504266629197,"score_gpt":0.2721111208052126,"score_spread":0.2597460781389206,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}