{"id":"W4385573045","doi":"10.18653/v1/2022.emnlp-main.115","title":"Enhancing Self-Consistency and Performance of Pre-Trained Language Models through Natural Language Inference","year":2022,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Canadian Institute for Advanced Research","keywords":"Computer science; Consistency (knowledge bases); Inference; Natural language; Natural language processing; Artificial intelligence; Language model; Natural (archaeology); Cognitive science; Psychology; History; Archaeology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01238318,0.001920375,0.00202613,0.001513103,0.0009526006,0.002702345,0.003878547,0.002476858,0.002434468],"category_scores_gemma":[0.04603836,0.001502331,0.001384863,0.0009668506,0.0009061642,0.007184919,0.003079286,0.004960336,0.003045694],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001299821,"about_ca_system_score_gemma":0.001905045,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01054285,"about_ca_topic_score_gemma":0.01775237,"domain_scores_codex":[0.9940469,0.00290758,0.0003438728,0.001793174,0.0005649798,0.0003434159],"domain_scores_gemma":[0.9612871,0.02842879,0.000880963,0.00552618,0.003338235,0.0005386372],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001760377,0.0009166639,0.01295211,0.0003693069,0.0008763258,0.0002105351,0.000541825,0.287949,0.01559787,0.003749704,0.01972768,0.6553487],"study_design_scores_gemma":[0.00006838047,0.00007369925,0.0006451593,0.00001814712,0.00007535594,0.00003809076,0.00004097305,0.9880541,0.005886723,0.00427971,0.0007976716,0.00002195938],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1536817,0.004254769,0.8010624,0.001696859,0.0008993337,0.0001942367,0.0009165565,0.03163955,0.0056546],"genre_scores_gemma":[0.7792966,0.000600205,0.2081565,0.0008941966,0.0004327722,0.000201111,0.003612399,0.002890843,0.003915381],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01238318,"threshold_uncertainty_score":0.06548929,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01190846749545949,"score_gpt":0.2489932632583665,"score_spread":0.2370847957629071,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}