{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IINCAV4AHJQGDGXMOS7YUZ23P6","short_pith_number":"pith:IINCAV4A","schema_version":"1.0","canonical_sha256":"421a2057803a60619aec74bf8a675b7f9d5da3b5bcec192672870df0992d4a09","source":{"kind":"arxiv","id":"2410.05451","version":3},"attestation_state":"computed","paper":{"title":"SecAlign: Defending Against Prompt Injection with Preference Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Arman Zharmagambetov, Chuan Guo, David Wagner, Kamalika Chaudhuri, Saeed Mahloujifar, Sizhe Chen","submitted_at":"2024-10-07T19:34:35Z","abstract_excerpt":"Large language models (LLMs) are becoming increasingly prevalent in modern software systems, interfacing between the user and the Internet to assist with tasks that require advanced language understanding. To accomplish these tasks, the LLM often uses external data sources such as user documents, web retrieval, results from API calls, etc. This opens up new avenues for attackers to manipulate the LLM via prompt injection. Adversarial prompts can be injected into external data sources to override the system's intended instruction and instead execute a malicious instruction. To mitigate this vul"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.05451","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2024-10-07T19:34:35Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"f8be96acd1a7b956458b89ac4a8aa7e4d3f798bae9df1b2dbe8b8a771129c32a","abstract_canon_sha256":"138321758175bae353f2b4480ccd5d6f4388b43cca4d0a50d3425d5b595ab230"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:31:02.651008Z","signature_b64":"8OSaUzxQnYQa15sbCnOW8gB+O2NV5z2+/cnpWEDnOg5l1fXfTWXpXYMI3LN1Pa3PXkPkiTNxmXTfrnNIT0JAAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"421a2057803a60619aec74bf8a675b7f9d5da3b5bcec192672870df0992d4a09","last_reissued_at":"2026-07-05T11:31:02.650489Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:31:02.650489Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SecAlign: Defending Against Prompt Injection with Preference Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Arman Zharmagambetov, Chuan Guo, David Wagner, Kamalika Chaudhuri, Saeed Mahloujifar, Sizhe Chen","submitted_at":"2024-10-07T19:34:35Z","abstract_excerpt":"Large language models (LLMs) are becoming increasingly prevalent in modern software systems, interfacing between the user and the Internet to assist with tasks that require advanced language understanding. To accomplish these tasks, the LLM often uses external data sources such as user documents, web retrieval, results from API calls, etc. This opens up new avenues for attackers to manipulate the LLM via prompt injection. Adversarial prompts can be injected into external data sources to override the system's intended instruction and instead execute a malicious instruction. To mitigate this vul"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.05451","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.05451/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.05451","created_at":"2026-07-05T11:31:02.650550+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.05451v3","created_at":"2026-07-05T11:31:02.650550+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.05451","created_at":"2026-07-05T11:31:02.650550+00:00"},{"alias_kind":"pith_short_12","alias_value":"IINCAV4AHJQG","created_at":"2026-07-05T11:31:02.650550+00:00"},{"alias_kind":"pith_short_16","alias_value":"IINCAV4AHJQGDGXM","created_at":"2026-07-05T11:31:02.650550+00:00"},{"alias_kind":"pith_short_8","alias_value":"IINCAV4A","created_at":"2026-07-05T11:31:02.650550+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02668","citing_title":"What You Approve Is What Executes: Consent Integrity for Black-Box LLM Agents","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30383","citing_title":"Whose Side Is Your Agent On? Multi-Party Principal Loyalty in LLM Agents","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28739","citing_title":"Agent Safety Is Action Alignment","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05206","citing_title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","ref_index":141,"is_internal_anchor":false},{"citing_arxiv_id":"2504.20472","citing_title":"Robustness via Referencing: Defending against Prompt Injection Attacks by Referencing the Executed Instruction","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2504.20984","citing_title":"ACE: A Security Architecture for LLM-Integrated App Systems","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2507.10610","citing_title":"LaSM: Layer-wise Scaling Mechanism for Defending Pop-up Attack on GUI Agents","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2510.23883","citing_title":"Agentic AI Security: Threats, Defenses, Evaluation, and Open Challenges","ref_index":180,"is_internal_anchor":false},{"citing_arxiv_id":"2512.00520","citing_title":"Toward a Safe Internet of Agents","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2504.19793","citing_title":"Prompt Injection Attack to Tool Selection in LLM Agents","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14290","citing_title":"Web Agents Should Adopt the Plan-Then-Execute Paradigm","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11868","citing_title":"IPI-proxy: An Intercepting Proxy for Red-Teaming Web-Browsing AI Agents Against Indirect Prompt Injection","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25109","citing_title":"Structured Security Auditing and Robustness Enhancement for Untrusted Agent Skills","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01078","citing_title":"A Sentence Relation-Based Approach to Sanitizing Malicious Instructions","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12817","citing_title":"Understanding and Improving Continuous Adversarial Training for LLMs via In-context Learning Theory","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IINCAV4AHJQGDGXMOS7YUZ23P6","json":"https://pith.science/pith/IINCAV4AHJQGDGXMOS7YUZ23P6.json","graph_json":"https://pith.science/api/pith-number/IINCAV4AHJQGDGXMOS7YUZ23P6/graph.json","events_json":"https://pith.science/api/pith-number/IINCAV4AHJQGDGXMOS7YUZ23P6/events.json","paper":"https://pith.science/paper/IINCAV4A"},"agent_actions":{"view_html":"https://pith.science/pith/IINCAV4AHJQGDGXMOS7YUZ23P6","download_json":"https://pith.science/pith/IINCAV4AHJQGDGXMOS7YUZ23P6.json","view_paper":"https://pith.science/paper/IINCAV4A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.05451&json=true","fetch_graph":"https://pith.science/api/pith-number/IINCAV4AHJQGDGXMOS7YUZ23P6/graph.json","fetch_events":"https://pith.science/api/pith-number/IINCAV4AHJQGDGXMOS7YUZ23P6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IINCAV4AHJQGDGXMOS7YUZ23P6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IINCAV4AHJQGDGXMOS7YUZ23P6/action/storage_attestation","attest_author":"https://pith.science/pith/IINCAV4AHJQGDGXMOS7YUZ23P6/action/author_attestation","sign_citation":"https://pith.science/pith/IINCAV4AHJQGDGXMOS7YUZ23P6/action/citation_signature","submit_replication":"https://pith.science/pith/IINCAV4AHJQGDGXMOS7YUZ23P6/action/replication_record"}},"created_at":"2026-07-05T11:31:02.650550+00:00","updated_at":"2026-07-05T11:31:02.650550+00:00"}