{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FQHN2T7K43GNX72GZ6XZJT4IYF","short_pith_number":"pith:FQHN2T7K","schema_version":"1.0","canonical_sha256":"2c0edd4feae6ccdbff46cfaf94cf88c154095834bd0289e78273757501661f7b","source":{"kind":"arxiv","id":"2312.14197","version":4},"attestation_state":"computed","paper":{"title":"Benchmarking and Defending Against Indirect Prompt Injection Attacks on Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bin Zhu, Emre Kiciman, Fangzhao Wu, Guangzhong Sun, Jingwei Yi, Xing Xie, Yueqi Xie","submitted_at":"2023-12-21T01:08:39Z","abstract_excerpt":"The integration of large language models with external content has enabled applications such as Microsoft Copilot but also introduced vulnerabilities to indirect prompt injection attacks. In these attacks, malicious instructions embedded within external content can manipulate LLM outputs, causing deviations from user expectations. To address this critical yet under-explored issue, we introduce the first benchmark for indirect prompt injection attacks, named BIPIA, to assess the risk of such vulnerabilities. Using BIPIA, we evaluate existing LLMs and find them universally vulnerable. Our analys"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.14197","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-12-21T01:08:39Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"792f3c75154a13698a61479941ecdcdc61df06431f81e0060938368e5e7a0199","abstract_canon_sha256":"67d9daac8f77bec732994a1a2db40e6cf121fd856a54b9d2552c988789ad46e7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:05:40.980746Z","signature_b64":"9OUuBR4suUEjd7O+AdkT9OxKdLKvbu/C1bEPEWvv2KMOO+f03dC5SIA7zW8JbBkL7D1w7+C+vNTaAM2WH57zDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c0edd4feae6ccdbff46cfaf94cf88c154095834bd0289e78273757501661f7b","last_reissued_at":"2026-07-05T10:05:40.980255Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:05:40.980255Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking and Defending Against Indirect Prompt Injection Attacks on Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bin Zhu, Emre Kiciman, Fangzhao Wu, Guangzhong Sun, Jingwei Yi, Xing Xie, Yueqi Xie","submitted_at":"2023-12-21T01:08:39Z","abstract_excerpt":"The integration of large language models with external content has enabled applications such as Microsoft Copilot but also introduced vulnerabilities to indirect prompt injection attacks. In these attacks, malicious instructions embedded within external content can manipulate LLM outputs, causing deviations from user expectations. To address this critical yet under-explored issue, we introduce the first benchmark for indirect prompt injection attacks, named BIPIA, to assess the risk of such vulnerabilities. Using BIPIA, we evaluate existing LLMs and find them universally vulnerable. Our analys"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.14197","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.14197/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.14197","created_at":"2026-07-05T10:05:40.980321+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.14197v4","created_at":"2026-07-05T10:05:40.980321+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.14197","created_at":"2026-07-05T10:05:40.980321+00:00"},{"alias_kind":"pith_short_12","alias_value":"FQHN2T7K43GN","created_at":"2026-07-05T10:05:40.980321+00:00"},{"alias_kind":"pith_short_16","alias_value":"FQHN2T7K43GNX72G","created_at":"2026-07-05T10:05:40.980321+00:00"},{"alias_kind":"pith_short_8","alias_value":"FQHN2T7K","created_at":"2026-07-05T10:05:40.980321+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":33,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22659","citing_title":"Confidently Wrong: Severity-Aware Calibration of Prompt-Injection Detectors under Attack Shift","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19660","citing_title":"A Layered Security Framework Against Prompt Injection in RAG-Based Chatbots","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18530","citing_title":"Evaluating Prompting-Based Defenses Against Domain-Camouflaged Injection Attacks","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10322","citing_title":"Game-Theoretic Multi-Agent Control for Robust Contextual Reasoning in LLMs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10304","citing_title":"MIRAGE: A Polarity-Flipping Encoding Subspace in LLM Agents","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09315","citing_title":"Brain-Prompt Injection: A Route-Safety Audit for BCI-LLM Agents","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08021","citing_title":"Semantic Quorum Assurance: Collective Certification for Non-Deterministic AI Infrastructure","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04141","citing_title":"Caught in the Act(ivation): Toward Pre-Output and Multi-Turn Detection of Credential Exfiltration by LLM Agents","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04109","citing_title":"Discourse-Role Labels as Presentation-Time Variables for Context Use in Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02959","citing_title":"Gate AI: LLM Security Benchmark Evaluation Methodology and Results","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30783","citing_title":"Security--Fidelity Tradeoffs: The Hidden Cost of Prompt Injection Defense","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31042","citing_title":"From Prompt Injection to Persistent Control: Defending Agentic Harness Against Trojan Backdoors","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05206","citing_title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2504.20472","citing_title":"Robustness via Referencing: Defending against Prompt Injection Attacks by Referencing the Executed Instruction","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19192","citing_title":"Hallucination as Exploit: Evidence-Carrying Multimodal Agents","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18133","citing_title":"An Empirical Study of Privacy Leakage Chains via Prompt Injection in Black-Box Chatbot Environments","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19192","citing_title":"Hallucination as Exploit: Evidence-Carrying Multimodal Agents","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2404.08144","citing_title":"LLM Agents can Autonomously Exploit One-day Vulnerabilities","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14290","citing_title":"Web Agents Should Adopt the Plan-Then-Execute Paradigm","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2403.14720","citing_title":"Defending Against Indirect Prompt Injection Attacks With Spotlighting","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23887","citing_title":"Evaluation of Prompt Injection Defenses in Large Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2406.13352","citing_title":"AgentDojo: A Dynamic Environment to Evaluate Prompt Injection Attacks and Defenses for LLM Agents","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11868","citing_title":"IPI-proxy: An Intercepting Proxy for Red-Teaming Web-Browsing AI Agents Against Indirect Prompt Injection","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2410.02644","citing_title":"Agent Security Bench (ASB): Formalizing and Benchmarking Attacks and Defenses in LLM-based Agents","ref_index":157,"is_internal_anchor":false},{"citing_arxiv_id":"2404.13208","citing_title":"The Instruction Hierarchy: Training LLMs to Prioritize Privileged Instructions","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FQHN2T7K43GNX72GZ6XZJT4IYF","json":"https://pith.science/pith/FQHN2T7K43GNX72GZ6XZJT4IYF.json","graph_json":"https://pith.science/api/pith-number/FQHN2T7K43GNX72GZ6XZJT4IYF/graph.json","events_json":"https://pith.science/api/pith-number/FQHN2T7K43GNX72GZ6XZJT4IYF/events.json","paper":"https://pith.science/paper/FQHN2T7K"},"agent_actions":{"view_html":"https://pith.science/pith/FQHN2T7K43GNX72GZ6XZJT4IYF","download_json":"https://pith.science/pith/FQHN2T7K43GNX72GZ6XZJT4IYF.json","view_paper":"https://pith.science/paper/FQHN2T7K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.14197&json=true","fetch_graph":"https://pith.science/api/pith-number/FQHN2T7K43GNX72GZ6XZJT4IYF/graph.json","fetch_events":"https://pith.science/api/pith-number/FQHN2T7K43GNX72GZ6XZJT4IYF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FQHN2T7K43GNX72GZ6XZJT4IYF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FQHN2T7K43GNX72GZ6XZJT4IYF/action/storage_attestation","attest_author":"https://pith.science/pith/FQHN2T7K43GNX72GZ6XZJT4IYF/action/author_attestation","sign_citation":"https://pith.science/pith/FQHN2T7K43GNX72GZ6XZJT4IYF/action/citation_signature","submit_replication":"https://pith.science/pith/FQHN2T7K43GNX72GZ6XZJT4IYF/action/replication_record"}},"created_at":"2026-07-05T10:05:40.980321+00:00","updated_at":"2026-07-05T10:05:40.980321+00:00"}