{"id":"W3101035550","doi":"","title":"Measuring Systematic Generalization in Neural Proof Generation with Transformers","year":2020,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Topic Modeling","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal; McGill University","funders":"","keywords":"Mathematical proof; Computer science; Backward chaining; Generalization; Chaining; Artificial intelligence; Natural deduction; Automated theorem proving; Inference; Theoretical computer science; Programming language; Mathematics; Inference engine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006694199,0.001115506,0.0006130201,0.001040624,0.000278985,0.001103836,0.001615624,0.001363978,0.001601872],"category_scores_gemma":[0.05228423,0.0006035756,0.0009938248,0.0006867726,0.001147759,0.00533431,0.00159398,0.00261706,0.0004656616],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001286137,"about_ca_system_score_gemma":0.001139792,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002924178,"about_ca_topic_score_gemma":0.004308581,"domain_scores_codex":[0.9974747,0.0009714987,0.0002478736,0.0006880756,0.0004550648,0.0001626368],"domain_scores_gemma":[0.9604189,0.03064331,0.002104006,0.004975502,0.00139482,0.000463497],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007949703,0.0005756917,0.03320811,0.0006651753,0.0005692081,0.0001897776,0.0005946931,0.7490464,0.01970432,0.006530267,0.001804043,0.1863174],"study_design_scores_gemma":[0.00003760072,0.0003585849,0.003465039,0.00002883032,0.00006994122,0.00009852624,0.00006396481,0.9708334,0.01128831,0.01335307,0.0003804451,0.00002228525],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7480198,0.0008109179,0.2434619,0.0006322694,0.00006670438,0.0002263747,0.0007725187,0.003264302,0.002745256],"genre_scores_gemma":[0.9488912,0.000203061,0.04901211,0.0001475153,0.00001310812,0.0001262868,0.001097059,0.0001033666,0.0004063666],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006694199,"threshold_uncertainty_score":0.03540272,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06149766231254278,"score_gpt":0.229747275178059,"score_spread":0.1682496128655162,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}