{"id":"W4412944158","doi":"10.18653/v1/2025.trl-1.20","title":"Sparks of Tabular Reasoning via Text2SQL Reinforcement Learning","year":2025,"lang":"en","type":"article","venue":"","topic":"Multi-Agent Systems and Negotiation","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute for Catastrophic Loss Reduction","keywords":"Reinforcement learning; Computer science; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003098783,0.00006961195,0.0001188227,0.00009908892,0.00007940878,0.00003858435,0.000263769,0.00003737677,0.0000638338],"category_scores_gemma":[0.00003835307,0.00006075249,0.00004163001,0.0002502061,0.000008686631,0.0001923556,0.0001348868,0.00006935753,0.00002386131],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002827934,"about_ca_system_score_gemma":0.00002852137,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002597055,"about_ca_topic_score_gemma":0.00000950206,"domain_scores_codex":[0.9992276,0.00004014002,0.0002521883,0.0001784976,0.0001695617,0.000131955],"domain_scores_gemma":[0.9994984,0.00003496931,0.0001103251,0.0002694107,0.00005946706,0.00002737788],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001497838,0.00009181562,0.03831178,0.0002570928,0.0001358066,0.000007469175,0.001971589,0.1028227,0.03602041,0.7286996,0.005550414,0.08611629],"study_design_scores_gemma":[0.0002834589,0.00003894428,0.005150401,0.0001118753,0.000005014439,8.827738e-7,0.00004727699,0.964941,0.01856103,0.0001435416,0.01062088,0.00009573375],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00774396,0.0000767057,0.9611504,0.0001260382,0.00023971,0.000117308,3.124843e-8,0.0000743739,0.03047145],"genre_scores_gemma":[0.9797061,0.000007617572,0.01152486,0.00008679905,0.00001654707,0.000005470581,0.000001484398,0.000002430043,0.00864866],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9719622,"threshold_uncertainty_score":0.2477415,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008045413565145598,"score_gpt":0.2406306985107835,"score_spread":0.2325852849456379,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}