{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MIR7ZKTJBJGFBF7XUSO2ZU7A4N","short_pith_number":"pith:MIR7ZKTJ","schema_version":"1.0","canonical_sha256":"6223fcaa690a4c5097f7a49dacd3e0e37660985de3fe674e20a294a3c01bdd84","source":{"kind":"arxiv","id":"2503.23157","version":2},"attestation_state":"computed","paper":{"title":"Reasoning-SQL: Reinforcement Learning with SQL Tailored Partial Rewards for Reasoning-Enhanced Text-to-SQL","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.DB","cs.PL"],"primary_cat":"cs.LG","authors_text":"Amin Saberi, Azalia Mirhoseini, Hailong Li, Mohammadreza Pourreza, Ruoxi Sun, Sercan \"O. Arik, Shayan Talaei, Xingchen Wan","submitted_at":"2025-03-29T17:29:30Z","abstract_excerpt":"Text-to-SQL is a challenging task involving multiple reasoning-intensive subtasks, including natural language understanding, database schema comprehension, and precise SQL query formulation. Existing approaches often rely on handcrafted reasoning paths with inductive biases that can limit their overall effectiveness. Motivated by the recent success of reasoning-enhanced models such as DeepSeek R1 and OpenAI o1, which effectively leverage reward-driven self-exploration to enhance reasoning capabilities and generalization, we propose a novel set of partial rewards tailored specifically for the T"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.23157","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2025-03-29T17:29:30Z","cross_cats_sorted":["cs.AI","cs.DB","cs.PL"],"title_canon_sha256":"0037b5fbaf9a3498b084a52f1ccd5cc9b7d89dd909ac7cb3cd1eba69a338b1ac","abstract_canon_sha256":"b7a2885f315c6b10e3aff8bb967fc315e3ed5f711fafea8bbb5cb91892d02baa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:42:45.861932Z","signature_b64":"MyBGs2w26C92BXBFSG6I0RSTineTTLtJ2RNN4yxn41vl4GVQtoNBoTtyUXWlwkAITD9Fjugqdi/2+8VlJ1XfCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6223fcaa690a4c5097f7a49dacd3e0e37660985de3fe674e20a294a3c01bdd84","last_reissued_at":"2026-07-05T10:42:45.861411Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:42:45.861411Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reasoning-SQL: Reinforcement Learning with SQL Tailored Partial Rewards for Reasoning-Enhanced Text-to-SQL","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.DB","cs.PL"],"primary_cat":"cs.LG","authors_text":"Amin Saberi, Azalia Mirhoseini, Hailong Li, Mohammadreza Pourreza, Ruoxi Sun, Sercan \"O. Arik, Shayan Talaei, Xingchen Wan","submitted_at":"2025-03-29T17:29:30Z","abstract_excerpt":"Text-to-SQL is a challenging task involving multiple reasoning-intensive subtasks, including natural language understanding, database schema comprehension, and precise SQL query formulation. Existing approaches often rely on handcrafted reasoning paths with inductive biases that can limit their overall effectiveness. Motivated by the recent success of reasoning-enhanced models such as DeepSeek R1 and OpenAI o1, which effectively leverage reward-driven self-exploration to enhance reasoning capabilities and generalization, we propose a novel set of partial rewards tailored specifically for the T"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.23157","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.23157/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.23157","created_at":"2026-07-05T10:42:45.861468+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.23157v2","created_at":"2026-07-05T10:42:45.861468+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.23157","created_at":"2026-07-05T10:42:45.861468+00:00"},{"alias_kind":"pith_short_12","alias_value":"MIR7ZKTJBJGF","created_at":"2026-07-05T10:42:45.861468+00:00"},{"alias_kind":"pith_short_16","alias_value":"MIR7ZKTJBJGFBF7X","created_at":"2026-07-05T10:42:45.861468+00:00"},{"alias_kind":"pith_short_8","alias_value":"MIR7ZKTJ","created_at":"2026-07-05T10:42:45.861468+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12387","citing_title":"TAHOE: Text-to-SQL with Automated Hint Optimization from Experience","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06825","citing_title":"Progress-SQL: Improving Reinforcement Learning for Text-to-SQL via Progressive Rewards","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23693","citing_title":"EXPO-SQL: Execution-based Clause-level Policy Optimization for Text-to-SQL","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2505.14174","citing_title":"Cheaper, Better, Faster, Stronger: Robust Text-to-SQL without Chain-of-Thought or Fine-Tuning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21792","citing_title":"Residual Skill Optimization for Text-to-SQL Ensembles","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2511.01008","citing_title":"MARS-SQL: A multi-agent reinforcement learning framework for Text-to-SQL","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12319","citing_title":"Data-aware candidate selection in NL2SQL translation via small separating instances","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03465","citing_title":"FINER-SQL: Boosting Small Language Models for Text-to-SQL","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04066","citing_title":"Adapt to Thrive! Adaptive Power-Mean Policy Optimization for Improved LLM Reasoning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04065","citing_title":"Free Energy-Driven Reinforcement Learning with Adaptive Advantage Shaping for Unsupervised Reasoning in LLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06804","citing_title":"LASER: A Data-Centric Method for Low-Cost and Efficient SQL Rewriting based on SQL-GRPO","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08057","citing_title":"CA-SQL: Complexity-Aware Inference Time Reasoning for Text-to-SQL via Exploration and Compute Budget Allocation","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MIR7ZKTJBJGFBF7XUSO2ZU7A4N","json":"https://pith.science/pith/MIR7ZKTJBJGFBF7XUSO2ZU7A4N.json","graph_json":"https://pith.science/api/pith-number/MIR7ZKTJBJGFBF7XUSO2ZU7A4N/graph.json","events_json":"https://pith.science/api/pith-number/MIR7ZKTJBJGFBF7XUSO2ZU7A4N/events.json","paper":"https://pith.science/paper/MIR7ZKTJ"},"agent_actions":{"view_html":"https://pith.science/pith/MIR7ZKTJBJGFBF7XUSO2ZU7A4N","download_json":"https://pith.science/pith/MIR7ZKTJBJGFBF7XUSO2ZU7A4N.json","view_paper":"https://pith.science/paper/MIR7ZKTJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.23157&json=true","fetch_graph":"https://pith.science/api/pith-number/MIR7ZKTJBJGFBF7XUSO2ZU7A4N/graph.json","fetch_events":"https://pith.science/api/pith-number/MIR7ZKTJBJGFBF7XUSO2ZU7A4N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MIR7ZKTJBJGFBF7XUSO2ZU7A4N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MIR7ZKTJBJGFBF7XUSO2ZU7A4N/action/storage_attestation","attest_author":"https://pith.science/pith/MIR7ZKTJBJGFBF7XUSO2ZU7A4N/action/author_attestation","sign_citation":"https://pith.science/pith/MIR7ZKTJBJGFBF7XUSO2ZU7A4N/action/citation_signature","submit_replication":"https://pith.science/pith/MIR7ZKTJBJGFBF7XUSO2ZU7A4N/action/replication_record"}},"created_at":"2026-07-05T10:42:45.861468+00:00","updated_at":"2026-07-05T10:42:45.861468+00:00"}