{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:EUZX52UUQA4C24ODQQFUW65EA4","short_pith_number":"pith:EUZX52UU","schema_version":"1.0","canonical_sha256":"25337eea9480382d71c3840b4b7ba4072d7b41d6d88989799889667328305ec3","source":{"kind":"arxiv","id":"2311.01173","version":1},"attestation_state":"computed","paper":{"title":"CRUSH4SQL: Collective Retrieval Using Schema Hallucination For Text2SQL","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dhruva Dhingra, Mayank Kothyari, Soumen Chakrabarti, Sunita Sarawagi","submitted_at":"2023-11-02T12:13:52Z","abstract_excerpt":"Existing Text-to-SQL generators require the entire schema to be encoded with the user text. This is expensive or impractical for large databases with tens of thousands of columns. Standard dense retrieval techniques are inadequate for schema subsetting of a large structured database, where the correct semantics of retrieval demands that we rank sets of schema elements rather than individual elements. In response, we propose a two-stage process for effective coverage during retrieval. First, we instruct an LLM to hallucinate a minimal DB schema deemed adequate to answer the query. We use the ha"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.01173","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-02T12:13:52Z","cross_cats_sorted":[],"title_canon_sha256":"f968dfea7637ff10de298b112750015081edcb079e4fc0fd1e102971da746e8e","abstract_canon_sha256":"d38b756b31420f238ec0e7a19e457658b5b4228755247929e44e3e24d5c0e5a8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:08:21.910121Z","signature_b64":"4go4pvd67G+HRKby1nVvbTSkjkob3tJEitpZsQIZMoIuybGL1+NKZjHc/1g/PHhRHgv2ZvROMIBdVIJ9msufAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"25337eea9480382d71c3840b4b7ba4072d7b41d6d88989799889667328305ec3","last_reissued_at":"2026-07-05T07:08:21.909660Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:08:21.909660Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CRUSH4SQL: Collective Retrieval Using Schema Hallucination For Text2SQL","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dhruva Dhingra, Mayank Kothyari, Soumen Chakrabarti, Sunita Sarawagi","submitted_at":"2023-11-02T12:13:52Z","abstract_excerpt":"Existing Text-to-SQL generators require the entire schema to be encoded with the user text. This is expensive or impractical for large databases with tens of thousands of columns. Standard dense retrieval techniques are inadequate for schema subsetting of a large structured database, where the correct semantics of retrieval demands that we rank sets of schema elements rather than individual elements. In response, we propose a two-stage process for effective coverage during retrieval. First, we instruct an LLM to hallucinate a minimal DB schema deemed adequate to answer the query. We use the ha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.01173","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.01173/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.01173","created_at":"2026-07-05T07:08:21.909717+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.01173v1","created_at":"2026-07-05T07:08:21.909717+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.01173","created_at":"2026-07-05T07:08:21.909717+00:00"},{"alias_kind":"pith_short_12","alias_value":"EUZX52UUQA4C","created_at":"2026-07-05T07:08:21.909717+00:00"},{"alias_kind":"pith_short_16","alias_value":"EUZX52UUQA4C24OD","created_at":"2026-07-05T07:08:21.909717+00:00"},{"alias_kind":"pith_short_8","alias_value":"EUZX52UU","created_at":"2026-07-05T07:08:21.909717+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.23104","citing_title":"RASL: Retrieval Augmented Schema Linking for Massive Database Text-to-SQL","ref_index":2023,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EUZX52UUQA4C24ODQQFUW65EA4","json":"https://pith.science/pith/EUZX52UUQA4C24ODQQFUW65EA4.json","graph_json":"https://pith.science/api/pith-number/EUZX52UUQA4C24ODQQFUW65EA4/graph.json","events_json":"https://pith.science/api/pith-number/EUZX52UUQA4C24ODQQFUW65EA4/events.json","paper":"https://pith.science/paper/EUZX52UU"},"agent_actions":{"view_html":"https://pith.science/pith/EUZX52UUQA4C24ODQQFUW65EA4","download_json":"https://pith.science/pith/EUZX52UUQA4C24ODQQFUW65EA4.json","view_paper":"https://pith.science/paper/EUZX52UU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.01173&json=true","fetch_graph":"https://pith.science/api/pith-number/EUZX52UUQA4C24ODQQFUW65EA4/graph.json","fetch_events":"https://pith.science/api/pith-number/EUZX52UUQA4C24ODQQFUW65EA4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EUZX52UUQA4C24ODQQFUW65EA4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EUZX52UUQA4C24ODQQFUW65EA4/action/storage_attestation","attest_author":"https://pith.science/pith/EUZX52UUQA4C24ODQQFUW65EA4/action/author_attestation","sign_citation":"https://pith.science/pith/EUZX52UUQA4C24ODQQFUW65EA4/action/citation_signature","submit_replication":"https://pith.science/pith/EUZX52UUQA4C24ODQQFUW65EA4/action/replication_record"}},"created_at":"2026-07-05T07:08:21.909717+00:00","updated_at":"2026-07-05T07:08:21.909717+00:00"}