{"id":"W4385572965","doi":"10.18653/v1/2022.emnlp-main.82","title":"Maieutic Prompting: Logically Consistent Reasoning with Recursive Explanations","year":2022,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":73,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Naval Information Warfare Center Pacific; Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency; Allen Institute for Artificial Intelligence","keywords":"Inference; Correctness; Computer science; Robustness (evolution); Commonsense reasoning; Artificial intelligence; Rule of inference; Machine learning; Natural language processing; Algorithm","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004149807,0.0015844,0.0006864965,0.0008492942,0.0005613659,0.001520002,0.00308826,0.002235481,0.01131154],"category_scores_gemma":[0.03033782,0.0006416188,0.001064697,0.0005347731,0.001268938,0.004027372,0.003001495,0.003831899,0.00305301],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009673536,"about_ca_system_score_gemma":0.002457383,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001890071,"about_ca_topic_score_gemma":0.00473851,"domain_scores_codex":[0.9973864,0.001202165,0.0001559482,0.0007261847,0.0004090988,0.0001201644],"domain_scores_gemma":[0.9824656,0.01310874,0.0005958735,0.002460534,0.001083658,0.0002854973],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001412107,0.0005651209,0.005420959,0.001862277,0.0001789672,0.0009652661,0.002470674,0.09901056,0.03683129,0.05901601,0.0563608,0.7359059],"study_design_scores_gemma":[0.0003246457,0.0002795772,0.0006795545,0.0001470439,0.00008112476,0.0003892073,0.0003159377,0.8224897,0.03514232,0.1101078,0.02996896,0.00007413204],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02186242,0.0004377567,0.9123029,0.001124006,0.0001538216,0.0003976696,0.001394534,0.05892584,0.003400997],"genre_scores_gemma":[0.2794943,0.0002160048,0.7108431,0.0009528208,0.0001155013,0.0003332687,0.002954408,0.001637456,0.00345316],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01131154,"threshold_uncertainty_score":0.03784084,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03000526500158486,"score_gpt":0.2339525538757103,"score_spread":0.2039472888741254,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}