{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:W2UL36UMVLJAT6ZNP6LXAWADPU","short_pith_number":"pith:W2UL36UM","schema_version":"1.0","canonical_sha256":"b6a8bdfa8caad209fb2d7f977058037d20c8a54ac9ab20c83c7d8c947449bf5c","source":{"kind":"arxiv","id":"2403.15879","version":6},"attestation_state":"computed","paper":{"title":"TrustSQL: Benchmarking Text-to-SQL Reliability with Penalty-Based Scoring","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Edward Choi, Gyubok Lee, Seonhee Cho, Woosog Chay","submitted_at":"2024-03-23T16:12:52Z","abstract_excerpt":"Text-to-SQL enables users to interact with databases using natural language, simplifying the retrieval and synthesis of information. Despite the remarkable success of large language models (LLMs) in translating natural language questions into SQL queries, widespread deployment remains limited due to two primary challenges. First, the effective use of text-to-SQL models depends on users' understanding of the model's capabilities-the scope of questions the model can correctly answer. Second, the absence of abstention mechanisms can lead to incorrect SQL generation going unnoticed, thereby underm"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.15879","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-03-23T16:12:52Z","cross_cats_sorted":[],"title_canon_sha256":"f6aaf556f206b6488ed6cf62a893280df9687e1dd6286a073458e9a5c04bc54d","abstract_canon_sha256":"a11882f4f0b2aadc240751c1c28672c3394bfff99e0df4e784fad431deb9c8cb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:38:58.568906Z","signature_b64":"tTVy6+TwBnX3UAbRXBVK8tak5XM/JcI69537vZGmpj/HYv0sm3IbEh3p8LgEEHz0QeJiXIoB3uR+dw3dtdWdCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b6a8bdfa8caad209fb2d7f977058037d20c8a54ac9ab20c83c7d8c947449bf5c","last_reissued_at":"2026-07-05T08:38:58.568285Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:38:58.568285Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TrustSQL: Benchmarking Text-to-SQL Reliability with Penalty-Based Scoring","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Edward Choi, Gyubok Lee, Seonhee Cho, Woosog Chay","submitted_at":"2024-03-23T16:12:52Z","abstract_excerpt":"Text-to-SQL enables users to interact with databases using natural language, simplifying the retrieval and synthesis of information. Despite the remarkable success of large language models (LLMs) in translating natural language questions into SQL queries, widespread deployment remains limited due to two primary challenges. First, the effective use of text-to-SQL models depends on users' understanding of the model's capabilities-the scope of questions the model can correctly answer. Second, the absence of abstention mechanisms can lead to incorrect SQL generation going unnoticed, thereby underm"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.15879","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.15879/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.15879","created_at":"2026-07-05T08:38:58.568351+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.15879v6","created_at":"2026-07-05T08:38:58.568351+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.15879","created_at":"2026-07-05T08:38:58.568351+00:00"},{"alias_kind":"pith_short_12","alias_value":"W2UL36UMVLJA","created_at":"2026-07-05T08:38:58.568351+00:00"},{"alias_kind":"pith_short_16","alias_value":"W2UL36UMVLJAT6ZN","created_at":"2026-07-05T08:38:58.568351+00:00"},{"alias_kind":"pith_short_8","alias_value":"W2UL36UM","created_at":"2026-07-05T08:38:58.568351+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W2UL36UMVLJAT6ZNP6LXAWADPU","json":"https://pith.science/pith/W2UL36UMVLJAT6ZNP6LXAWADPU.json","graph_json":"https://pith.science/api/pith-number/W2UL36UMVLJAT6ZNP6LXAWADPU/graph.json","events_json":"https://pith.science/api/pith-number/W2UL36UMVLJAT6ZNP6LXAWADPU/events.json","paper":"https://pith.science/paper/W2UL36UM"},"agent_actions":{"view_html":"https://pith.science/pith/W2UL36UMVLJAT6ZNP6LXAWADPU","download_json":"https://pith.science/pith/W2UL36UMVLJAT6ZNP6LXAWADPU.json","view_paper":"https://pith.science/paper/W2UL36UM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.15879&json=true","fetch_graph":"https://pith.science/api/pith-number/W2UL36UMVLJAT6ZNP6LXAWADPU/graph.json","fetch_events":"https://pith.science/api/pith-number/W2UL36UMVLJAT6ZNP6LXAWADPU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W2UL36UMVLJAT6ZNP6LXAWADPU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W2UL36UMVLJAT6ZNP6LXAWADPU/action/storage_attestation","attest_author":"https://pith.science/pith/W2UL36UMVLJAT6ZNP6LXAWADPU/action/author_attestation","sign_citation":"https://pith.science/pith/W2UL36UMVLJAT6ZNP6LXAWADPU/action/citation_signature","submit_replication":"https://pith.science/pith/W2UL36UMVLJAT6ZNP6LXAWADPU/action/replication_record"}},"created_at":"2026-07-05T08:38:58.568351+00:00","updated_at":"2026-07-05T08:38:58.568351+00:00"}