{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YIT2SJPV37O3OH6NOTKLY5O5GX","short_pith_number":"pith:YIT2SJPV","schema_version":"1.0","canonical_sha256":"c227a925f5dfddb71fcd74d4bc75dd35d62c34ece3e84ba904bd9ebda9c63a41","source":{"kind":"arxiv","id":"2411.01705","version":2},"attestation_state":"computed","paper":{"title":"Data Extraction Attacks in Retrieval-Augmented Generation via Backdoors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CR","authors_text":"Amir Houmansadr, Hong Yu, Junda Wang, Yuefeng Peng","submitted_at":"2024-11-03T22:27:40Z","abstract_excerpt":"Despite significant advancements, large language models (LLMs) still struggle with providing accurate answers when lacking domain-specific or up-to-date knowledge. Retrieval-Augmented Generation (RAG) addresses this limitation by incorporating external knowledge bases, but it also introduces new attack surfaces. In this paper, we investigate data extraction attacks targeting RAG's knowledge databases. We show that previous prompt injection-based extraction attacks largely rely on the instruction-following capabilities of LLMs. As a result, they fail on models that are less responsive to such m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.01705","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-11-03T22:27:40Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"8f741665824ce02e6079d54690a39f3b25770af7d3fda4f23dab965a8f947522","abstract_canon_sha256":"78223ede1401c8fc1b0231e620a758fad1361a33868af74bf8badf2c333edca8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:16.840863Z","signature_b64":"Qa26MXPrMMdQlGX1uzVbfCX7lx+r2k2ELjFXJc3VZQvAxLgEDIHyy9i4FSKXdGGGIkl/nkxdRUyvF5OqIcGKDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c227a925f5dfddb71fcd74d4bc75dd35d62c34ece3e84ba904bd9ebda9c63a41","last_reissued_at":"2026-07-05T10:41:16.840357Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:16.840357Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Data Extraction Attacks in Retrieval-Augmented Generation via Backdoors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CR","authors_text":"Amir Houmansadr, Hong Yu, Junda Wang, Yuefeng Peng","submitted_at":"2024-11-03T22:27:40Z","abstract_excerpt":"Despite significant advancements, large language models (LLMs) still struggle with providing accurate answers when lacking domain-specific or up-to-date knowledge. Retrieval-Augmented Generation (RAG) addresses this limitation by incorporating external knowledge bases, but it also introduces new attack surfaces. In this paper, we investigate data extraction attacks targeting RAG's knowledge databases. We show that previous prompt injection-based extraction attacks largely rely on the instruction-following capabilities of LLMs. As a result, they fail on models that are less responsive to such m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.01705","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.01705/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.01705","created_at":"2026-07-05T10:41:16.840421+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.01705v2","created_at":"2026-07-05T10:41:16.840421+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.01705","created_at":"2026-07-05T10:41:16.840421+00:00"},{"alias_kind":"pith_short_12","alias_value":"YIT2SJPV37O3","created_at":"2026-07-05T10:41:16.840421+00:00"},{"alias_kind":"pith_short_16","alias_value":"YIT2SJPV37O3OH6N","created_at":"2026-07-05T10:41:16.840421+00:00"},{"alias_kind":"pith_short_8","alias_value":"YIT2SJPV","created_at":"2026-07-05T10:41:16.840421+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10091","citing_title":"SoK: Colluding Adversaries in Machine Learning Pipelines","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2510.06719","citing_title":"Differentially Private Synthetic Text Generation for Retrieval-Augmented Generation (RAG)","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YIT2SJPV37O3OH6NOTKLY5O5GX","json":"https://pith.science/pith/YIT2SJPV37O3OH6NOTKLY5O5GX.json","graph_json":"https://pith.science/api/pith-number/YIT2SJPV37O3OH6NOTKLY5O5GX/graph.json","events_json":"https://pith.science/api/pith-number/YIT2SJPV37O3OH6NOTKLY5O5GX/events.json","paper":"https://pith.science/paper/YIT2SJPV"},"agent_actions":{"view_html":"https://pith.science/pith/YIT2SJPV37O3OH6NOTKLY5O5GX","download_json":"https://pith.science/pith/YIT2SJPV37O3OH6NOTKLY5O5GX.json","view_paper":"https://pith.science/paper/YIT2SJPV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.01705&json=true","fetch_graph":"https://pith.science/api/pith-number/YIT2SJPV37O3OH6NOTKLY5O5GX/graph.json","fetch_events":"https://pith.science/api/pith-number/YIT2SJPV37O3OH6NOTKLY5O5GX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YIT2SJPV37O3OH6NOTKLY5O5GX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YIT2SJPV37O3OH6NOTKLY5O5GX/action/storage_attestation","attest_author":"https://pith.science/pith/YIT2SJPV37O3OH6NOTKLY5O5GX/action/author_attestation","sign_citation":"https://pith.science/pith/YIT2SJPV37O3OH6NOTKLY5O5GX/action/citation_signature","submit_replication":"https://pith.science/pith/YIT2SJPV37O3OH6NOTKLY5O5GX/action/replication_record"}},"created_at":"2026-07-05T10:41:16.840421+00:00","updated_at":"2026-07-05T10:41:16.840421+00:00"}