{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XRK6LBD5HDVWBNE6G7U2OJ4LUO","short_pith_number":"pith:XRK6LBD5","schema_version":"1.0","canonical_sha256":"bc55e5847d38eb60b49e37e9a7278ba3a47a2804b0dd4b538ee471d383323d3a","source":{"kind":"arxiv","id":"2402.12566","version":3},"attestation_state":"computed","paper":{"title":"GenAudit: Fixing Factual Errors in Language Model Outputs with Evidence","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Byron C. Wallace, Jeffrey P. Bigham, Kundan Krishna, Prakhar Gupta, Sanjana Ramprasad, Zachary C. Lipton","submitted_at":"2024-02-19T21:45:55Z","abstract_excerpt":"LLMs can generate factually incorrect statements even when provided access to reference documents. Such errors can be dangerous in high-stakes applications (e.g., document-grounded QA for healthcare or finance). We present GenAudit -- a tool intended to assist fact-checking LLM responses for document-grounded tasks. GenAudit suggests edits to the LLM response by revising or removing claims that are not supported by the reference document, and also presents evidence from the reference for facts that do appear to have support. We train models to execute these tasks, and design an interactive int"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.12566","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-19T21:45:55Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"ce04ecc36864cef8f286f471764460fec8297e3bbec227633b8309239a2ad85b","abstract_canon_sha256":"e04797616dded11262713274a6682c7f10fadec48bf9c6f4795a720d6c966340"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:02:43.799597Z","signature_b64":"sQahZa/rtObGvitBSeglG8/dZ9ZxrEAWvFcXMNWTfwUM9Zu3nZixbijHLK+bBggRG2WuP3htI+viHttoOYgFDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bc55e5847d38eb60b49e37e9a7278ba3a47a2804b0dd4b538ee471d383323d3a","last_reissued_at":"2026-07-05T10:02:43.799118Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:02:43.799118Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GenAudit: Fixing Factual Errors in Language Model Outputs with Evidence","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Byron C. Wallace, Jeffrey P. Bigham, Kundan Krishna, Prakhar Gupta, Sanjana Ramprasad, Zachary C. Lipton","submitted_at":"2024-02-19T21:45:55Z","abstract_excerpt":"LLMs can generate factually incorrect statements even when provided access to reference documents. Such errors can be dangerous in high-stakes applications (e.g., document-grounded QA for healthcare or finance). We present GenAudit -- a tool intended to assist fact-checking LLM responses for document-grounded tasks. GenAudit suggests edits to the LLM response by revising or removing claims that are not supported by the reference document, and also presents evidence from the reference for facts that do appear to have support. We train models to execute these tasks, and design an interactive int"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.12566","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.12566/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.12566","created_at":"2026-07-05T10:02:43.799173+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.12566v3","created_at":"2026-07-05T10:02:43.799173+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.12566","created_at":"2026-07-05T10:02:43.799173+00:00"},{"alias_kind":"pith_short_12","alias_value":"XRK6LBD5HDVW","created_at":"2026-07-05T10:02:43.799173+00:00"},{"alias_kind":"pith_short_16","alias_value":"XRK6LBD5HDVWBNE6","created_at":"2026-07-05T10:02:43.799173+00:00"},{"alias_kind":"pith_short_8","alias_value":"XRK6LBD5","created_at":"2026-07-05T10:02:43.799173+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.00361","citing_title":"Attribution Gradients: Incrementally Unfolding Citations for Critical Examination of Attributed AI Answers","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08082","citing_title":"From Binary Groundedness to Support Relations: Towards a Reader-Centred Taxonomy for Comprehension of AI Output","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08898","citing_title":"Omakase: proactive assistance with actionable suggestions for evolving scientific research projects","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XRK6LBD5HDVWBNE6G7U2OJ4LUO","json":"https://pith.science/pith/XRK6LBD5HDVWBNE6G7U2OJ4LUO.json","graph_json":"https://pith.science/api/pith-number/XRK6LBD5HDVWBNE6G7U2OJ4LUO/graph.json","events_json":"https://pith.science/api/pith-number/XRK6LBD5HDVWBNE6G7U2OJ4LUO/events.json","paper":"https://pith.science/paper/XRK6LBD5"},"agent_actions":{"view_html":"https://pith.science/pith/XRK6LBD5HDVWBNE6G7U2OJ4LUO","download_json":"https://pith.science/pith/XRK6LBD5HDVWBNE6G7U2OJ4LUO.json","view_paper":"https://pith.science/paper/XRK6LBD5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.12566&json=true","fetch_graph":"https://pith.science/api/pith-number/XRK6LBD5HDVWBNE6G7U2OJ4LUO/graph.json","fetch_events":"https://pith.science/api/pith-number/XRK6LBD5HDVWBNE6G7U2OJ4LUO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XRK6LBD5HDVWBNE6G7U2OJ4LUO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XRK6LBD5HDVWBNE6G7U2OJ4LUO/action/storage_attestation","attest_author":"https://pith.science/pith/XRK6LBD5HDVWBNE6G7U2OJ4LUO/action/author_attestation","sign_citation":"https://pith.science/pith/XRK6LBD5HDVWBNE6G7U2OJ4LUO/action/citation_signature","submit_replication":"https://pith.science/pith/XRK6LBD5HDVWBNE6G7U2OJ4LUO/action/replication_record"}},"created_at":"2026-07-05T10:02:43.799173+00:00","updated_at":"2026-07-05T10:02:43.799173+00:00"}