{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:HPP3DKCASMPQ62LMCSSFCAYYBY","short_pith_number":"pith:HPP3DKCA","schema_version":"1.0","canonical_sha256":"3bdfb1a840931f0f696c14a45103180e230f2ef31a128a3cf2a1dd17d765cdad","source":{"kind":"arxiv","id":"2608.06539","version":1},"attestation_state":"computed","paper":{"title":"Don't `Well, Actually' Me Unless You Know What You're Talking About: Weak Presupposition Verification Degrades General QA Performance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hila Gonen, Shenran Wang, Vered Shwartz","submitted_at":"2026-08-06T19:43:10Z","abstract_excerpt":"False-presupposition QA (FPQA) tests LLMs on their ability to identify false presuppositions in questions and abstain or correct them rather than reinforcing false assumptions. The common approach reduces the task to prompting LLMs to extract presuppositions and fact checking each presupposition. While the performance on dedicated benchmarks keeps improving, evaluation largely focuses on questions with false presuppositions (FPQs) while ignoring the performance on ``normal'' questions (TPQs). Since many benchmarks over-represent FPQs compared to their natural occurrence, the result is that per"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.06539","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-08-06T19:43:10Z","cross_cats_sorted":[],"title_canon_sha256":"562a5b2a1d7ea72cf14561119b8a33ca2870e820c99d40629a87ce7a03c22445","abstract_canon_sha256":"83f76d87f4fd8b53c16223734db5a0f650d5ba442299f326dc271e9fb075f927"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-10T01:10:58.437497Z","signature_b64":"1w0HcsHmQYLUEnVHlOOqda+ydQvjaxZj9pp6SSH2RBxvWPxdWy6JaRZgWaM82Wn4DBZuSfhSvsnYA/9IlXm+DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3bdfb1a840931f0f696c14a45103180e230f2ef31a128a3cf2a1dd17d765cdad","last_reissued_at":"2026-08-10T01:10:58.429783Z","signature_status":"signed_v1","first_computed_at":"2026-08-10T01:10:58.429783Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Don't `Well, Actually' Me Unless You Know What You're Talking About: Weak Presupposition Verification Degrades General QA Performance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hila Gonen, Shenran Wang, Vered Shwartz","submitted_at":"2026-08-06T19:43:10Z","abstract_excerpt":"False-presupposition QA (FPQA) tests LLMs on their ability to identify false presuppositions in questions and abstain or correct them rather than reinforcing false assumptions. The common approach reduces the task to prompting LLMs to extract presuppositions and fact checking each presupposition. While the performance on dedicated benchmarks keeps improving, evaluation largely focuses on questions with false presuppositions (FPQs) while ignoring the performance on ``normal'' questions (TPQs). Since many benchmarks over-represent FPQs compared to their natural occurrence, the result is that per"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.06539","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.06539/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.06539","created_at":"2026-08-10T01:10:58.434049+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.06539v1","created_at":"2026-08-10T01:10:58.434049+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.06539","created_at":"2026-08-10T01:10:58.434049+00:00"},{"alias_kind":"pith_short_12","alias_value":"HPP3DKCASMPQ","created_at":"2026-08-10T01:10:58.434049+00:00"},{"alias_kind":"pith_short_16","alias_value":"HPP3DKCASMPQ62LM","created_at":"2026-08-10T01:10:58.434049+00:00"},{"alias_kind":"pith_short_8","alias_value":"HPP3DKCA","created_at":"2026-08-10T01:10:58.434049+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HPP3DKCASMPQ62LMCSSFCAYYBY","json":"https://pith.science/pith/HPP3DKCASMPQ62LMCSSFCAYYBY.json","graph_json":"https://pith.science/api/pith-number/HPP3DKCASMPQ62LMCSSFCAYYBY/graph.json","events_json":"https://pith.science/api/pith-number/HPP3DKCASMPQ62LMCSSFCAYYBY/events.json","paper":"https://pith.science/paper/HPP3DKCA"},"agent_actions":{"view_html":"https://pith.science/pith/HPP3DKCASMPQ62LMCSSFCAYYBY","download_json":"https://pith.science/pith/HPP3DKCASMPQ62LMCSSFCAYYBY.json","view_paper":"https://pith.science/paper/HPP3DKCA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.06539&json=true","fetch_graph":"https://pith.science/api/pith-number/HPP3DKCASMPQ62LMCSSFCAYYBY/graph.json","fetch_events":"https://pith.science/api/pith-number/HPP3DKCASMPQ62LMCSSFCAYYBY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HPP3DKCASMPQ62LMCSSFCAYYBY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HPP3DKCASMPQ62LMCSSFCAYYBY/action/storage_attestation","attest_author":"https://pith.science/pith/HPP3DKCASMPQ62LMCSSFCAYYBY/action/author_attestation","sign_citation":"https://pith.science/pith/HPP3DKCASMPQ62LMCSSFCAYYBY/action/citation_signature","submit_replication":"https://pith.science/pith/HPP3DKCASMPQ62LMCSSFCAYYBY/action/replication_record"}},"created_at":"2026-08-10T01:10:58.434049+00:00","updated_at":"2026-08-10T01:10:58.434049+00:00"}