{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3NRKBP46X6S62I7JLINOU743FD","short_pith_number":"pith:3NRKBP46","schema_version":"1.0","canonical_sha256":"db62a0bf9ebfa5ed23e95a1aea7f9b28f752b819cdf90206b6bb0b24bc75c7fd","source":{"kind":"arxiv","id":"2411.14110","version":2},"attestation_state":"computed","paper":{"title":"Feedback-Guided Extraction of Knowledge Base from Retrieval-Augmented LLM Applications","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Changyue Jiang, Chenfu Bao, Geng Hong, Min Yang, Xudong Pan, Yang Chen","submitted_at":"2024-11-21T13:18:03Z","abstract_excerpt":"Retrieval-Augmented Generation (RAG) expands the knowledge boundary of large language models (LLMs) by integrating external knowledge bases, whose construction is often time-consuming and laborious. If an adversary extracts the knowledge base verbatim, it not only severely infringes the owner's intellectual property but also enables the adversary to replicate the application's functionality for unfair competition. Previous works on knowledge base extraction are limited either by low extraction coverage (usually less than 4%) in query-based attacks or by impractical assumptions of white-box acc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.14110","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CR","submitted_at":"2024-11-21T13:18:03Z","cross_cats_sorted":[],"title_canon_sha256":"238e8556c075b3800dca07c4d3d4179a7eb3c72d55e59a7c1fcf9b0b4f239893","abstract_canon_sha256":"b550cf36cc90dff094b12c39e0e815ad9582cc520a784c12ab00fe67ea254d2e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:50:29.116465Z","signature_b64":"gDknqXRejWoZ/q0imF4zYih1OAWe8ZPboZWU7oUjYnGobPhbwXAOFKUzFieMPm6PRSQto4Evooe18AGgkE6YDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"db62a0bf9ebfa5ed23e95a1aea7f9b28f752b819cdf90206b6bb0b24bc75c7fd","last_reissued_at":"2026-07-05T11:50:29.116021Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:50:29.116021Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Feedback-Guided Extraction of Knowledge Base from Retrieval-Augmented LLM Applications","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Changyue Jiang, Chenfu Bao, Geng Hong, Min Yang, Xudong Pan, Yang Chen","submitted_at":"2024-11-21T13:18:03Z","abstract_excerpt":"Retrieval-Augmented Generation (RAG) expands the knowledge boundary of large language models (LLMs) by integrating external knowledge bases, whose construction is often time-consuming and laborious. If an adversary extracts the knowledge base verbatim, it not only severely infringes the owner's intellectual property but also enables the adversary to replicate the application's functionality for unfair competition. Previous works on knowledge base extraction are limited either by low extraction coverage (usually less than 4%) in query-based attacks or by impractical assumptions of white-box acc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.14110","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.14110/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.14110","created_at":"2026-07-05T11:50:29.116069+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.14110v2","created_at":"2026-07-05T11:50:29.116069+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.14110","created_at":"2026-07-05T11:50:29.116069+00:00"},{"alias_kind":"pith_short_12","alias_value":"3NRKBP46X6S6","created_at":"2026-07-05T11:50:29.116069+00:00"},{"alias_kind":"pith_short_16","alias_value":"3NRKBP46X6S62I7J","created_at":"2026-07-05T11:50:29.116069+00:00"},{"alias_kind":"pith_short_8","alias_value":"3NRKBP46","created_at":"2026-07-05T11:50:29.116069+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26627","citing_title":"Agents That Know Too Much: A Data-Centric Survey of Privacy in LLM Agents","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26197","citing_title":"Hierarchical Long-Term Semantic Memory for LinkedIn's Hiring Agent","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2504.02181","citing_title":"A Survey of Scaling in Large Language Model Reasoning","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18762","citing_title":"ALDEN: Boosting Private Data Extraction from Retrieval-Augmented Generation Systems via Active Learning and Distribution Estimation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09033","citing_title":"ShadowMerge: A Novel Poisoning Attack on Graph-Based Agent Memory via Relation-Channel Conflicts","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09033","citing_title":"ShadowMerge: A Novel Poisoning Attack on Graph-Based Agent Memory via Relation-Channel Conflicts","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26197","citing_title":"Hierarchical Long-Term Semantic Memory for LinkedIn's Hiring Agent","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09033","citing_title":"ShadowMerge: A Novel Poisoning Attack on Graph-Based Agent Memory via Relation-Channel Conflicts","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10253","citing_title":"Knowledge Poisoning Attacks on Medical Multi-Modal Retrieval-Augmented Generation","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3NRKBP46X6S62I7JLINOU743FD","json":"https://pith.science/pith/3NRKBP46X6S62I7JLINOU743FD.json","graph_json":"https://pith.science/api/pith-number/3NRKBP46X6S62I7JLINOU743FD/graph.json","events_json":"https://pith.science/api/pith-number/3NRKBP46X6S62I7JLINOU743FD/events.json","paper":"https://pith.science/paper/3NRKBP46"},"agent_actions":{"view_html":"https://pith.science/pith/3NRKBP46X6S62I7JLINOU743FD","download_json":"https://pith.science/pith/3NRKBP46X6S62I7JLINOU743FD.json","view_paper":"https://pith.science/paper/3NRKBP46","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.14110&json=true","fetch_graph":"https://pith.science/api/pith-number/3NRKBP46X6S62I7JLINOU743FD/graph.json","fetch_events":"https://pith.science/api/pith-number/3NRKBP46X6S62I7JLINOU743FD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3NRKBP46X6S62I7JLINOU743FD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3NRKBP46X6S62I7JLINOU743FD/action/storage_attestation","attest_author":"https://pith.science/pith/3NRKBP46X6S62I7JLINOU743FD/action/author_attestation","sign_citation":"https://pith.science/pith/3NRKBP46X6S62I7JLINOU743FD/action/citation_signature","submit_replication":"https://pith.science/pith/3NRKBP46X6S62I7JLINOU743FD/action/replication_record"}},"created_at":"2026-07-05T11:50:29.116069+00:00","updated_at":"2026-07-05T11:50:29.116069+00:00"}