{"id":"W4412944158","doi":"10.18653/v1/2025.trl-1.20","title":"Sparks of Tabular Reasoning via Text2SQL Reinforcement Learning","year":2025,"lang":"en","type":"article","venue":"","topic":"Multi-Agent Systems and Negotiation","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute for Catastrophic Loss Reduction","keywords":"Reinforcement learning; Computer science; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003326145,0.00100566,0.0007846667,0.0005313499,0.0005240791,0.001419347,0.00349155,0.001034062,0.006251447],"category_scores_gemma":[0.01392078,0.0005110633,0.0008499155,0.0005155757,0.001255684,0.002830992,0.002626067,0.002799713,0.001772366],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001506992,"about_ca_system_score_gemma":0.002447712,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006626125,"about_ca_topic_score_gemma":0.01141587,"domain_scores_codex":[0.9982049,0.0006710382,0.0001026579,0.0004856777,0.0003937043,0.0001418464],"domain_scores_gemma":[0.9944615,0.003091114,0.0002793582,0.001273377,0.0005939712,0.0003007235],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009011822,0.001354352,0.006247045,0.0004925074,0.000137562,0.0003559798,0.0006989553,0.6169863,0.009765223,0.03089268,0.01970058,0.3124677],"study_design_scores_gemma":[0.00006025725,0.00004365194,0.0001101087,0.00000773156,0.000006403517,0.00001225459,0.00003476431,0.9845349,0.002032588,0.01166035,0.001490714,0.00000640226],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1400805,0.0005245025,0.8131163,0.001952722,0.0002394896,0.0004438691,0.001517821,0.03404202,0.008082855],"genre_scores_gemma":[0.658923,0.000147818,0.3337404,0.0007130197,0.00005972217,0.0003192828,0.002417039,0.0007698649,0.002909841],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006626125,"threshold_uncertainty_score":0.02091318,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008045413565145598,"score_gpt":0.2406306985107835,"score_spread":0.2325852849456379,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}