{"id":"W4414596238","doi":"10.1007/978-3-032-04614-7_24","title":"MATATA: Weakly Supervised End-to-End MAthematical Tool-Augmented Reasoning for Tabular Applications","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Merck Canada Inc. (Canada); University of Toronto","funders":"","keywords":"Planner; Commonsense reasoning; Automated reasoning; Outcome (game theory); Model-based reasoning; Language model; Non-monotonic logic; Reasoning system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001769344,0.0018874,0.0007917054,0.0007522018,0.0006648429,0.001754234,0.004115701,0.001634916,0.009270102],"category_scores_gemma":[0.006666967,0.0008865488,0.001755834,0.0005498221,0.0009385913,0.002817958,0.003404966,0.003614548,0.005598273],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001114746,"about_ca_system_score_gemma":0.002581622,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0056065,"about_ca_topic_score_gemma":0.01338754,"domain_scores_codex":[0.9987128,0.000378249,0.00008454474,0.0004549406,0.0002786529,0.00009074444],"domain_scores_gemma":[0.9974343,0.001172068,0.0001446238,0.0007041762,0.00042282,0.0001219996],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007026993,0.0006943874,0.002701622,0.0008081693,0.0002569715,0.0003313401,0.0007611491,0.2659305,0.02118389,0.01345328,0.05327177,0.6399042],"study_design_scores_gemma":[0.00004467292,0.00007619693,0.0001915231,0.00002698399,0.00002027285,0.00004235743,0.00005949654,0.9699424,0.008401887,0.01267732,0.00850048,0.00001635729],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01120642,0.0002295804,0.9354082,0.0001757225,0.00009133647,0.0003197736,0.001013385,0.04810531,0.003450204],"genre_scores_gemma":[0.1318731,0.0001175883,0.8523911,0.0003081365,0.00003215373,0.0005812243,0.005990885,0.001847178,0.006858762],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009270102,"threshold_uncertainty_score":0.03101158,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01252736102412455,"score_gpt":0.2740569173297306,"score_spread":0.261529556305606,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}