{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YZC5K3PLCIYCUJTKFC4BPHYZED","short_pith_number":"pith:YZC5K3PL","schema_version":"1.0","canonical_sha256":"c645d56deb12302a266a28b8179f1920ee926277223f5268725bd450613de61f","source":{"kind":"arxiv","id":"2502.03233","version":1},"attestation_state":"computed","paper":{"title":"Exploring the Security Threats of Knowledge Base Poisoning in Retrieval-Augmented Code Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CR","authors_text":"Bo Lin, Liqian Chen, Shangwen Wang, Xiaoguang Mao","submitted_at":"2025-02-05T14:49:12Z","abstract_excerpt":"The integration of Large Language Models (LLMs) into software development has revolutionized the field, particularly through the use of Retrieval-Augmented Code Generation (RACG) systems that enhance code generation with information from external knowledge bases. However, the security implications of RACG systems, particularly the risks posed by vulnerable code examples in the knowledge base, remain largely unexplored. This risk is particularly concerning given that public code repositories, which often serve as the sources for knowledge base collection in RACG systems, are usually accessible "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.03233","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2025-02-05T14:49:12Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"64775d8e02eb56c82cdb0e894c53f467f7328d6f7a8adc8f82503050a410bdef","abstract_canon_sha256":"2cabc4a6c96dee749e344d2a04b6a2368515bd6fdcf620872fd35c5278fd52ad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:10:01.830782Z","signature_b64":"3oiGXKxQObE/SBgDi2HtCccJT4+V2wmB/BL48GjV/Nt7n4Mx9T2E76PZlc4/dRQH6LDXJWGhYjodKpF9+OzUAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c645d56deb12302a266a28b8179f1920ee926277223f5268725bd450613de61f","last_reissued_at":"2026-07-05T10:10:01.830276Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:10:01.830276Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring the Security Threats of Knowledge Base Poisoning in Retrieval-Augmented Code Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CR","authors_text":"Bo Lin, Liqian Chen, Shangwen Wang, Xiaoguang Mao","submitted_at":"2025-02-05T14:49:12Z","abstract_excerpt":"The integration of Large Language Models (LLMs) into software development has revolutionized the field, particularly through the use of Retrieval-Augmented Code Generation (RACG) systems that enhance code generation with information from external knowledge bases. However, the security implications of RACG systems, particularly the risks posed by vulnerable code examples in the knowledge base, remain largely unexplored. This risk is particularly concerning given that public code repositories, which often serve as the sources for knowledge base collection in RACG systems, are usually accessible "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.03233","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.03233/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.03233","created_at":"2026-07-05T10:10:01.830346+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.03233v1","created_at":"2026-07-05T10:10:01.830346+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.03233","created_at":"2026-07-05T10:10:01.830346+00:00"},{"alias_kind":"pith_short_12","alias_value":"YZC5K3PLCIYC","created_at":"2026-07-05T10:10:01.830346+00:00"},{"alias_kind":"pith_short_16","alias_value":"YZC5K3PLCIYCUJTK","created_at":"2026-07-05T10:10:01.830346+00:00"},{"alias_kind":"pith_short_8","alias_value":"YZC5K3PL","created_at":"2026-07-05T10:10:01.830346+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01280","citing_title":"Fixed-Set Robustness in Programming by Example: Example Corruption and Semantic Partition Recovery","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2506.03535","citing_title":"Across Programming Language Silos: A Study on Cross-Lingual Retrieval-augmented Code Generation","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YZC5K3PLCIYCUJTKFC4BPHYZED","json":"https://pith.science/pith/YZC5K3PLCIYCUJTKFC4BPHYZED.json","graph_json":"https://pith.science/api/pith-number/YZC5K3PLCIYCUJTKFC4BPHYZED/graph.json","events_json":"https://pith.science/api/pith-number/YZC5K3PLCIYCUJTKFC4BPHYZED/events.json","paper":"https://pith.science/paper/YZC5K3PL"},"agent_actions":{"view_html":"https://pith.science/pith/YZC5K3PLCIYCUJTKFC4BPHYZED","download_json":"https://pith.science/pith/YZC5K3PLCIYCUJTKFC4BPHYZED.json","view_paper":"https://pith.science/paper/YZC5K3PL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.03233&json=true","fetch_graph":"https://pith.science/api/pith-number/YZC5K3PLCIYCUJTKFC4BPHYZED/graph.json","fetch_events":"https://pith.science/api/pith-number/YZC5K3PLCIYCUJTKFC4BPHYZED/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YZC5K3PLCIYCUJTKFC4BPHYZED/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YZC5K3PLCIYCUJTKFC4BPHYZED/action/storage_attestation","attest_author":"https://pith.science/pith/YZC5K3PLCIYCUJTKFC4BPHYZED/action/author_attestation","sign_citation":"https://pith.science/pith/YZC5K3PLCIYCUJTKFC4BPHYZED/action/citation_signature","submit_replication":"https://pith.science/pith/YZC5K3PLCIYCUJTKFC4BPHYZED/action/replication_record"}},"created_at":"2026-07-05T10:10:01.830346+00:00","updated_at":"2026-07-05T10:10:01.830346+00:00"}