{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XX5BUXU7YRPCVJQHP3CZQVSPBX","short_pith_number":"pith:XX5BUXU7","schema_version":"1.0","canonical_sha256":"bdfa1a5e9fc45e2aa6077ec598564f0de0d223df508a291be5a05314d35a5973","source":{"kind":"arxiv","id":"2408.12733","version":2},"attestation_state":"computed","paper":{"title":"SQL-GEN: Bridging the Dialect Gap for Text-to-SQL Via Synthetic Data And Model Merging","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL","cs.DB","cs.LG"],"primary_cat":"cs.AI","authors_text":"Hailong Li, Lesly Miculicich, Mohammadreza Pourreza, Ruoxi Sun, Sercan O. Arik, Tomas Pfister","submitted_at":"2024-08-22T20:50:48Z","abstract_excerpt":"Recent advances in Text-to-SQL have largely focused on the SQLite dialect, neglecting the diverse landscape of SQL dialects like BigQuery and PostgreSQL. This limitation is due to the diversity in SQL syntaxes and functions, along with the high cost of collecting and curating SQL-specific training data. To address this, we introduce SQL-GEN, a framework for generating high-quality synthetic training data for any SQL dialect, guided by readily available dialect-specific tutorials. SQL-GEN significantly improves cross-dialect Text-to-SQL performance, boosting execution accuracy by up to 20\\% ove"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.12733","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.AI","submitted_at":"2024-08-22T20:50:48Z","cross_cats_sorted":["cs.CL","cs.DB","cs.LG"],"title_canon_sha256":"4e2c5a2786a951a3827aa8d338101eb262ca7b48ebebd1d3838d491ad88f8591","abstract_canon_sha256":"9b32a287ff38819bfbc1924044e2b95b86dfb5f81de5abe42315a2b9b19f6bcc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:14:55.299398Z","signature_b64":"PB3ebP63ncviFfKyQoCnYSqbYswRVOz6/r0yRbqvA8mMWFMgiYZW/3sBF5fD5xbCXglE56HIgO/0ps4W7MbMAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bdfa1a5e9fc45e2aa6077ec598564f0de0d223df508a291be5a05314d35a5973","last_reissued_at":"2026-07-05T09:14:55.298905Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:14:55.298905Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SQL-GEN: Bridging the Dialect Gap for Text-to-SQL Via Synthetic Data And Model Merging","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL","cs.DB","cs.LG"],"primary_cat":"cs.AI","authors_text":"Hailong Li, Lesly Miculicich, Mohammadreza Pourreza, Ruoxi Sun, Sercan O. Arik, Tomas Pfister","submitted_at":"2024-08-22T20:50:48Z","abstract_excerpt":"Recent advances in Text-to-SQL have largely focused on the SQLite dialect, neglecting the diverse landscape of SQL dialects like BigQuery and PostgreSQL. This limitation is due to the diversity in SQL syntaxes and functions, along with the high cost of collecting and curating SQL-specific training data. To address this, we introduce SQL-GEN, a framework for generating high-quality synthetic training data for any SQL dialect, guided by readily available dialect-specific tutorials. SQL-GEN significantly improves cross-dialect Text-to-SQL performance, boosting execution accuracy by up to 20\\% ove"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.12733","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.12733/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.12733","created_at":"2026-07-05T09:14:55.298964+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.12733v2","created_at":"2026-07-05T09:14:55.298964+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.12733","created_at":"2026-07-05T09:14:55.298964+00:00"},{"alias_kind":"pith_short_12","alias_value":"XX5BUXU7YRPC","created_at":"2026-07-05T09:14:55.298964+00:00"},{"alias_kind":"pith_short_16","alias_value":"XX5BUXU7YRPCVJQH","created_at":"2026-07-05T09:14:55.298964+00:00"},{"alias_kind":"pith_short_8","alias_value":"XX5BUXU7","created_at":"2026-07-05T09:14:55.298964+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.22843","citing_title":"Knowledge Distillation for Low-Resource Open-source Text-to-SQL Model","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22313","citing_title":"CLARITY: A Framework and Benchmark for Conversational Language Ambiguity and Unanswerability in Interactive NL2SQL Systems","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04066","citing_title":"Adapt to Thrive! Adaptive Power-Mean Policy Optimization for Improved LLM Reasoning","ref_index":150,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04065","citing_title":"Free Energy-Driven Reinforcement Learning with Adaptive Advantage Shaping for Unsupervised Reasoning in LLMs","ref_index":165,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07796","citing_title":"PolySQL: Scaling Text-to-SQL Evaluation Across SQL Dialects via Automated Backend Isomorphism","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XX5BUXU7YRPCVJQHP3CZQVSPBX","json":"https://pith.science/pith/XX5BUXU7YRPCVJQHP3CZQVSPBX.json","graph_json":"https://pith.science/api/pith-number/XX5BUXU7YRPCVJQHP3CZQVSPBX/graph.json","events_json":"https://pith.science/api/pith-number/XX5BUXU7YRPCVJQHP3CZQVSPBX/events.json","paper":"https://pith.science/paper/XX5BUXU7"},"agent_actions":{"view_html":"https://pith.science/pith/XX5BUXU7YRPCVJQHP3CZQVSPBX","download_json":"https://pith.science/pith/XX5BUXU7YRPCVJQHP3CZQVSPBX.json","view_paper":"https://pith.science/paper/XX5BUXU7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.12733&json=true","fetch_graph":"https://pith.science/api/pith-number/XX5BUXU7YRPCVJQHP3CZQVSPBX/graph.json","fetch_events":"https://pith.science/api/pith-number/XX5BUXU7YRPCVJQHP3CZQVSPBX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XX5BUXU7YRPCVJQHP3CZQVSPBX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XX5BUXU7YRPCVJQHP3CZQVSPBX/action/storage_attestation","attest_author":"https://pith.science/pith/XX5BUXU7YRPCVJQHP3CZQVSPBX/action/author_attestation","sign_citation":"https://pith.science/pith/XX5BUXU7YRPCVJQHP3CZQVSPBX/action/citation_signature","submit_replication":"https://pith.science/pith/XX5BUXU7YRPCVJQHP3CZQVSPBX/action/replication_record"}},"created_at":"2026-07-05T09:14:55.298964+00:00","updated_at":"2026-07-05T09:14:55.298964+00:00"}