{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:4YPZXN5RZ4ELSE7RL2OYW7SNUW","short_pith_number":"pith:4YPZXN5R","schema_version":"1.0","canonical_sha256":"e61f9bb7b1cf08b913f15e9d8b7e4da58c04311aa5cff63120081169d273a09a","source":{"kind":"arxiv","id":"2607.20537","version":1},"attestation_state":"computed","paper":{"title":"ReliableTableQA:How Much Supervision Does Reliability Annotation Need?","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Hsin-Tai Wu, Huei-Chung Hu, Koyo Kobayashi","submitted_at":"2026-07-10T16:57:30Z","abstract_excerpt":"We introduce ReliableTableQA, a framework for training an LLM to annotate the statistical reliability of tabular QA results, not whether the query is answerable, but whether the computed answer is statistically meaningful. In real enterprise analytics, a syntactically correct SQL query can return a value that is based on too small a sample, has an excessively wide confidence interval, or is too confounded to support action. Existing systems answer confidently in all such cases, a failure we quantify as the Unreliable Confident Answer Rate (UCAR). We contribute (1) a ten-category reliability ta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.20537","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-10T16:57:30Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e05a9c7568d0823cf3964c017748d60a71fcac27a855b016601f9dad59e6da7c","abstract_canon_sha256":"a858fbe2d22e657f0c8c74b93532db3ec9d0d877c24168ad06784fd6bd69a434"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-24T00:23:21.239121Z","signature_b64":"kmHVS+lU8KpTrd3rKDgt2hT+uJ4aSZKfuLLI48n6xFQKPX3IDY9e797Oa8DXrV9vQfllcZN97I5f/rbqpRU+AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e61f9bb7b1cf08b913f15e9d8b7e4da58c04311aa5cff63120081169d273a09a","last_reissued_at":"2026-07-24T00:23:21.238188Z","signature_status":"signed_v1","first_computed_at":"2026-07-24T00:23:21.238188Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReliableTableQA:How Much Supervision Does Reliability Annotation Need?","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Hsin-Tai Wu, Huei-Chung Hu, Koyo Kobayashi","submitted_at":"2026-07-10T16:57:30Z","abstract_excerpt":"We introduce ReliableTableQA, a framework for training an LLM to annotate the statistical reliability of tabular QA results, not whether the query is answerable, but whether the computed answer is statistically meaningful. In real enterprise analytics, a syntactically correct SQL query can return a value that is based on too small a sample, has an excessively wide confidence interval, or is too confounded to support action. Existing systems answer confidently in all such cases, a failure we quantify as the Unreliable Confident Answer Rate (UCAR). We contribute (1) a ten-category reliability ta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.20537","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.20537/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.20537","created_at":"2026-07-24T00:23:21.238682+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.20537v1","created_at":"2026-07-24T00:23:21.238682+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.20537","created_at":"2026-07-24T00:23:21.238682+00:00"},{"alias_kind":"pith_short_12","alias_value":"4YPZXN5RZ4EL","created_at":"2026-07-24T00:23:21.238682+00:00"},{"alias_kind":"pith_short_16","alias_value":"4YPZXN5RZ4ELSE7R","created_at":"2026-07-24T00:23:21.238682+00:00"},{"alias_kind":"pith_short_8","alias_value":"4YPZXN5R","created_at":"2026-07-24T00:23:21.238682+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4YPZXN5RZ4ELSE7RL2OYW7SNUW","json":"https://pith.science/pith/4YPZXN5RZ4ELSE7RL2OYW7SNUW.json","graph_json":"https://pith.science/api/pith-number/4YPZXN5RZ4ELSE7RL2OYW7SNUW/graph.json","events_json":"https://pith.science/api/pith-number/4YPZXN5RZ4ELSE7RL2OYW7SNUW/events.json","paper":"https://pith.science/paper/4YPZXN5R"},"agent_actions":{"view_html":"https://pith.science/pith/4YPZXN5RZ4ELSE7RL2OYW7SNUW","download_json":"https://pith.science/pith/4YPZXN5RZ4ELSE7RL2OYW7SNUW.json","view_paper":"https://pith.science/paper/4YPZXN5R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.20537&json=true","fetch_graph":"https://pith.science/api/pith-number/4YPZXN5RZ4ELSE7RL2OYW7SNUW/graph.json","fetch_events":"https://pith.science/api/pith-number/4YPZXN5RZ4ELSE7RL2OYW7SNUW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4YPZXN5RZ4ELSE7RL2OYW7SNUW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4YPZXN5RZ4ELSE7RL2OYW7SNUW/action/storage_attestation","attest_author":"https://pith.science/pith/4YPZXN5RZ4ELSE7RL2OYW7SNUW/action/author_attestation","sign_citation":"https://pith.science/pith/4YPZXN5RZ4ELSE7RL2OYW7SNUW/action/citation_signature","submit_replication":"https://pith.science/pith/4YPZXN5RZ4ELSE7RL2OYW7SNUW/action/replication_record"}},"created_at":"2026-07-24T00:23:21.238682+00:00","updated_at":"2026-07-24T00:23:21.238682+00:00"}