{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:2CHH2LDHX3L7MCRXJP7F3ZT27E","short_pith_number":"pith:2CHH2LDH","schema_version":"1.0","canonical_sha256":"d08e7d2c67bed7f60a374bfe5de67af93dda2e5f676f96f926b4db2dee4f8d39","source":{"kind":"arxiv","id":"2606.23667","version":1},"attestation_state":"computed","paper":{"title":"The Table Says Otherwise: Testing LLMs with Counterfactual Relational Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DB","authors_text":"Chunwei Liu, Xinzhi Wang","submitted_at":"2026-06-22T17:52:14Z","abstract_excerpt":"Large language models (LLMs) are increasingly used to answer natural-language questions over structured data. However, when a table contains familiar real-world facts, it is unclear whether the model answers by reading the provided data or by recalling knowledge learned during pretraining. This distinction is important for database applications, where the provided tables should be the source of truth. In this paper, we introduce ContraTable, a paired original-counterfactual benchmark for evaluating whether LLMs ground their answers in relational tables. We build the benchmark with two aligned "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2606.23667","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DB","submitted_at":"2026-06-22T17:52:14Z","cross_cats_sorted":[],"title_canon_sha256":"326d19822ccd84f6f4de2604cc07031646eb4d10c727c84f6fab5dc8828e88e1","abstract_canon_sha256":"5449e0d575d3f939907ed9f22cd923e4a59c80615d992a2a4b977cbfa31737bf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-23T03:14:34.120845Z","signature_b64":"yOTraCNIC2vFUw+q7vpjrJEJeoLzT9Dzch8qcoqO16pWxKFNBjIHSFLx3q7jxhiVdxz10a/118zjePXnfIfYAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d08e7d2c67bed7f60a374bfe5de67af93dda2e5f676f96f926b4db2dee4f8d39","last_reissued_at":"2026-06-23T03:14:34.120422Z","signature_status":"signed_v1","first_computed_at":"2026-06-23T03:14:34.120422Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Table Says Otherwise: Testing LLMs with Counterfactual Relational Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DB","authors_text":"Chunwei Liu, Xinzhi Wang","submitted_at":"2026-06-22T17:52:14Z","abstract_excerpt":"Large language models (LLMs) are increasingly used to answer natural-language questions over structured data. However, when a table contains familiar real-world facts, it is unclear whether the model answers by reading the provided data or by recalling knowledge learned during pretraining. This distinction is important for database applications, where the provided tables should be the source of truth. In this paper, we introduce ContraTable, a paired original-counterfactual benchmark for evaluating whether LLMs ground their answers in relational tables. We build the benchmark with two aligned "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2606.23667","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2606.23667/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2606.23667","created_at":"2026-06-23T03:14:34.120482+00:00"},{"alias_kind":"arxiv_version","alias_value":"2606.23667v1","created_at":"2026-06-23T03:14:34.120482+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2606.23667","created_at":"2026-06-23T03:14:34.120482+00:00"},{"alias_kind":"pith_short_12","alias_value":"2CHH2LDHX3L7","created_at":"2026-06-23T03:14:34.120482+00:00"},{"alias_kind":"pith_short_16","alias_value":"2CHH2LDHX3L7MCRX","created_at":"2026-06-23T03:14:34.120482+00:00"},{"alias_kind":"pith_short_8","alias_value":"2CHH2LDH","created_at":"2026-06-23T03:14:34.120482+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2CHH2LDHX3L7MCRXJP7F3ZT27E","json":"https://pith.science/pith/2CHH2LDHX3L7MCRXJP7F3ZT27E.json","graph_json":"https://pith.science/api/pith-number/2CHH2LDHX3L7MCRXJP7F3ZT27E/graph.json","events_json":"https://pith.science/api/pith-number/2CHH2LDHX3L7MCRXJP7F3ZT27E/events.json","paper":"https://pith.science/paper/2CHH2LDH"},"agent_actions":{"view_html":"https://pith.science/pith/2CHH2LDHX3L7MCRXJP7F3ZT27E","download_json":"https://pith.science/pith/2CHH2LDHX3L7MCRXJP7F3ZT27E.json","view_paper":"https://pith.science/paper/2CHH2LDH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2606.23667&json=true","fetch_graph":"https://pith.science/api/pith-number/2CHH2LDHX3L7MCRXJP7F3ZT27E/graph.json","fetch_events":"https://pith.science/api/pith-number/2CHH2LDHX3L7MCRXJP7F3ZT27E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2CHH2LDHX3L7MCRXJP7F3ZT27E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2CHH2LDHX3L7MCRXJP7F3ZT27E/action/storage_attestation","attest_author":"https://pith.science/pith/2CHH2LDHX3L7MCRXJP7F3ZT27E/action/author_attestation","sign_citation":"https://pith.science/pith/2CHH2LDHX3L7MCRXJP7F3ZT27E/action/citation_signature","submit_replication":"https://pith.science/pith/2CHH2LDHX3L7MCRXJP7F3ZT27E/action/replication_record"}},"created_at":"2026-06-23T03:14:34.120482+00:00","updated_at":"2026-06-23T03:14:34.120482+00:00"}