{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:L6D4C732PLL4ZFA54E4Z454NB4","short_pith_number":"pith:L6D4C732","canonical_record":{"source":{"id":"2406.13352","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-06-19T08:55:56Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"1c8176bcec8e2e4c59ecdc117c198c7d4bb960b68ef087eef061e001b1c802f7","abstract_canon_sha256":"48164cd8bf85271696bdf5efb2ba6afb3dafe96b2b7b4e9a371d857387748971"},"schema_version":"1.0"},"canonical_sha256":"5f87c17f7a7ad7cc941de1399e778d0f051b0d4de487a14b8cf099a9dec360e8","source":{"kind":"arxiv","id":"2406.13352","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.13352","created_at":"2026-07-05T09:39:56Z"},{"alias_kind":"arxiv_version","alias_value":"2406.13352v3","created_at":"2026-07-05T09:39:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.13352","created_at":"2026-07-05T09:39:56Z"},{"alias_kind":"pith_short_12","alias_value":"L6D4C732PLL4","created_at":"2026-07-05T09:39:56Z"},{"alias_kind":"pith_short_16","alias_value":"L6D4C732PLL4ZFA5","created_at":"2026-07-05T09:39:56Z"},{"alias_kind":"pith_short_8","alias_value":"L6D4C732","created_at":"2026-07-05T09:39:56Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:L6D4C732PLL4ZFA54E4Z454NB4","target":"record","payload":{"canonical_record":{"source":{"id":"2406.13352","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-06-19T08:55:56Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"1c8176bcec8e2e4c59ecdc117c198c7d4bb960b68ef087eef061e001b1c802f7","abstract_canon_sha256":"48164cd8bf85271696bdf5efb2ba6afb3dafe96b2b7b4e9a371d857387748971"},"schema_version":"1.0"},"canonical_sha256":"5f87c17f7a7ad7cc941de1399e778d0f051b0d4de487a14b8cf099a9dec360e8","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:39:56.114651Z","signature_b64":"wm4QeWhZ/2q5vlR+558mSiqD5g6APbZ/gESmjW8PXUJyELtEuctbHJKWwAoMWuOXftmbR+2TBhx6fAk9/zDOCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5f87c17f7a7ad7cc941de1399e778d0f051b0d4de487a14b8cf099a9dec360e8","last_reissued_at":"2026-07-05T09:39:56.114097Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:39:56.114097Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2406.13352","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:39:56Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"esaJp085CLfzmZmnbo+kFufGeJuBv1i5H6AN1F7jFCVoAC4pwCQ2rBX+AOGEVZn9bInbU+tproTFFbAI0l0/DA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T19:24:31.780594Z"},"content_sha256":"2a07ab63b992e968affed581c64c154e4174ec74b2e314e97b1b662e1137c661","schema_version":"1.0","event_id":"sha256:2a07ab63b992e968affed581c64c154e4174ec74b2e314e97b1b662e1137c661"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:L6D4C732PLL4ZFA54E4Z454NB4","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"AgentDojo: A Dynamic Environment to Evaluate Prompt Injection Attacks and Defenses for LLM Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"AgentDojo introduces an extensible evaluation framework populated with realistic agent tasks and security test cases to measure prompt injection robustness in tool-using LLM agents.","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Edoardo Debenedetti, Florian Tram\\`er, Jie Zhang, Luca Beurer-Kellner, Marc Fischer, Mislav Balunovi\\'c","submitted_at":"2024-06-19T08:55:56Z","abstract_excerpt":"AI agents aim to solve complex tasks by combining text-based reasoning with external tool calls. Unfortunately, AI agents are vulnerable to prompt injection attacks where data returned by external tools hijacks the agent to execute malicious tasks. To measure the adversarial robustness of AI agents, we introduce AgentDojo, an evaluation framework for agents that execute tools over untrusted data. To capture the evolving nature of attacks and defenses, AgentDojo is not a static test suite, but rather an extensible environment for designing and evaluating new agent tasks, defenses, and adaptive "},"claims":{"count":3,"items":[{"kind":"strongest_claim","text":"To measure the adversarial robustness of AI agents, we introduce AgentDojo, an evaluation framework for agents that execute tools over untrusted data... state-of-the-art LLMs fail at many tasks (even in the absence of attacks), and existing prompt injection attacks break some security properties but not all.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The 97 tasks and 629 security test cases in AgentDojo accurately represent real-world agent behaviors and the space of prompt injection threats from untrusted tool outputs.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"AgentDojo introduces an extensible evaluation framework populated with realistic agent tasks and security test cases to measure prompt injection robustness in tool-using LLM agents.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"}],"snapshot_sha256":"032492c6190497bf7aa2519621033555d4da71eae1fb0294f5f9e87f55daec7b"},"source":{"id":"2406.13352","kind":"arxiv","version":3},"verdict":{"id":"4fc49c92-5692-497c-bbfd-36e636bda5c3","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-13T06:32:13.330618Z","strongest_claim":"To measure the adversarial robustness of AI agents, we introduce AgentDojo, an evaluation framework for agents that execute tools over untrusted data... state-of-the-art LLMs fail at many tasks (even in the absence of attacks), and existing prompt injection attacks break some security properties but not all.","one_line_summary":"AgentDojo introduces an extensible evaluation framework populated with realistic agent tasks and security test cases to measure prompt injection robustness in tool-using LLM agents.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The 97 tasks and 629 security test cases in AgentDojo accurately represent real-world agent behaviors and the space of prompt injection threats from untrusted tool outputs.","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.13352/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":79,"sample":[{"doi":"10.1145/3650203.3663326","year":2024,"title":"Croissant: A metadata format for ml-ready datasets","work_id":"b13e2013-4762-4e9a-97b5-74aa550ddbde","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2024,"title":"The Claude 3 Model Family: Opus, Sonnet, Haiku","work_id":"d692169b-0d64-4638-b78a-14576f1ca990","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2024,"title":"Tool use (function calling)","work_id":"60a265a5-bb85-404d-826e-7c8d6445462a","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2022,"title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","work_id":"a1f2574b-a899-4713-be60-c87ba332656c","ref_index":4,"cited_arxiv_id":"2204.05862","is_internal_anchor":true},{"doi":"","year":2020,"title":"Language models are few-shot learners","work_id":"0668d8ab-e260-41af-ace5-c065a5d6ed0f","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":79,"snapshot_sha256":"244d136a52f46838e7e4f400c972039790fee701e12e44f8c7123b9a4315b4e6","internal_anchors":21},"formal_canon":{"evidence_count":2,"snapshot_sha256":"6650ed14c6d804793b910a86af22aa0705bdd4b5d85fbdebe3f71f8943deedcc"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"4fc49c92-5692-497c-bbfd-36e636bda5c3"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:39:56Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"I2rV3YtyWVZ9PUnDsOpzwEECaU+g3jtO9tPc3XcR/gQXAplapNf0U/DavGuBHOiQv7b5bwLttLODYcFWe0fPBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T19:24:31.781696Z"},"content_sha256":"55078a5ed2b4a22be9aabfb23d7a59b229e9352f48e1a417b9e99006a7d750ca","schema_version":"1.0","event_id":"sha256:55078a5ed2b4a22be9aabfb23d7a59b229e9352f48e1a417b9e99006a7d750ca"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/L6D4C732PLL4ZFA54E4Z454NB4/bundle.json","state_url":"https://pith.science/pith/L6D4C732PLL4ZFA54E4Z454NB4/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/L6D4C732PLL4ZFA54E4Z454NB4/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T19:24:31Z","links":{"resolver":"https://pith.science/pith/L6D4C732PLL4ZFA54E4Z454NB4","bundle":"https://pith.science/pith/L6D4C732PLL4ZFA54E4Z454NB4/bundle.json","state":"https://pith.science/pith/L6D4C732PLL4ZFA54E4Z454NB4/state.json","well_known_bundle":"https://pith.science/.well-known/pith/L6D4C732PLL4ZFA54E4Z454NB4/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:L6D4C732PLL4ZFA54E4Z454NB4","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"48164cd8bf85271696bdf5efb2ba6afb3dafe96b2b7b4e9a371d857387748971","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-06-19T08:55:56Z","title_canon_sha256":"1c8176bcec8e2e4c59ecdc117c198c7d4bb960b68ef087eef061e001b1c802f7"},"schema_version":"1.0","source":{"id":"2406.13352","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.13352","created_at":"2026-07-05T09:39:56Z"},{"alias_kind":"arxiv_version","alias_value":"2406.13352v3","created_at":"2026-07-05T09:39:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.13352","created_at":"2026-07-05T09:39:56Z"},{"alias_kind":"pith_short_12","alias_value":"L6D4C732PLL4","created_at":"2026-07-05T09:39:56Z"},{"alias_kind":"pith_short_16","alias_value":"L6D4C732PLL4ZFA5","created_at":"2026-07-05T09:39:56Z"},{"alias_kind":"pith_short_8","alias_value":"L6D4C732","created_at":"2026-07-05T09:39:56Z"}],"graph_snapshots":[{"event_id":"sha256:55078a5ed2b4a22be9aabfb23d7a59b229e9352f48e1a417b9e99006a7d750ca","target":"graph","created_at":"2026-07-05T09:39:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":3,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"To measure the adversarial robustness of AI agents, we introduce AgentDojo, an evaluation framework for agents that execute tools over untrusted data... state-of-the-art LLMs fail at many tasks (even in the absence of attacks), and existing prompt injection attacks break some security properties but not all."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The 97 tasks and 629 security test cases in AgentDojo accurately represent real-world agent behaviors and the space of prompt injection threats from untrusted tool outputs."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"AgentDojo introduces an extensible evaluation framework populated with realistic agent tasks and security test cases to measure prompt injection robustness in tool-using LLM agents."}],"snapshot_sha256":"032492c6190497bf7aa2519621033555d4da71eae1fb0294f5f9e87f55daec7b"},"formal_canon":{"evidence_count":2,"snapshot_sha256":"6650ed14c6d804793b910a86af22aa0705bdd4b5d85fbdebe3f71f8943deedcc"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2406.13352/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"AI agents aim to solve complex tasks by combining text-based reasoning with external tool calls. Unfortunately, AI agents are vulnerable to prompt injection attacks where data returned by external tools hijacks the agent to execute malicious tasks. To measure the adversarial robustness of AI agents, we introduce AgentDojo, an evaluation framework for agents that execute tools over untrusted data. To capture the evolving nature of attacks and defenses, AgentDojo is not a static test suite, but rather an extensible environment for designing and evaluating new agent tasks, defenses, and adaptive ","authors_text":"Edoardo Debenedetti, Florian Tram\\`er, Jie Zhang, Luca Beurer-Kellner, Marc Fischer, Mislav Balunovi\\'c","cross_cats":["cs.LG"],"headline":"AgentDojo introduces an extensible evaluation framework populated with realistic agent tasks and security test cases to measure prompt injection robustness in tool-using LLM agents.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-06-19T08:55:56Z","title":"AgentDojo: A Dynamic Environment to Evaluate Prompt Injection Attacks and Defenses for LLM Agents"},"references":{"count":79,"internal_anchors":21,"resolved_work":79,"sample":[{"cited_arxiv_id":"","doi":"10.1145/3650203.3663326","is_internal_anchor":false,"ref_index":1,"title":"Croissant: A metadata format for ml-ready datasets","work_id":"b13e2013-4762-4e9a-97b5-74aa550ddbde","year":2024},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":2,"title":"The Claude 3 Model Family: Opus, Sonnet, Haiku","work_id":"d692169b-0d64-4638-b78a-14576f1ca990","year":2024},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":3,"title":"Tool use (function calling)","work_id":"60a265a5-bb85-404d-826e-7c8d6445462a","year":2024},{"cited_arxiv_id":"2204.05862","doi":"","is_internal_anchor":true,"ref_index":4,"title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","work_id":"a1f2574b-a899-4713-be60-c87ba332656c","year":2022},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":5,"title":"Language models are few-shot learners","work_id":"0668d8ab-e260-41af-ace5-c065a5d6ed0f","year":2020}],"snapshot_sha256":"244d136a52f46838e7e4f400c972039790fee701e12e44f8c7123b9a4315b4e6"},"source":{"id":"2406.13352","kind":"arxiv","version":3},"verdict":{"created_at":"2026-05-13T06:32:13.330618Z","id":"4fc49c92-5692-497c-bbfd-36e636bda5c3","model_set":{"reader":"grok-4.3"},"one_line_summary":"AgentDojo introduces an extensible evaluation framework populated with realistic agent tasks and security test cases to measure prompt injection robustness in tool-using LLM agents.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"","strongest_claim":"To measure the adversarial robustness of AI agents, we introduce AgentDojo, an evaluation framework for agents that execute tools over untrusted data... state-of-the-art LLMs fail at many tasks (even in the absence of attacks), and existing prompt injection attacks break some security properties but not all.","weakest_assumption":"The 97 tasks and 629 security test cases in AgentDojo accurately represent real-world agent behaviors and the space of prompt injection threats from untrusted tool outputs."}},"verdict_id":"4fc49c92-5692-497c-bbfd-36e636bda5c3"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:2a07ab63b992e968affed581c64c154e4174ec74b2e314e97b1b662e1137c661","target":"record","created_at":"2026-07-05T09:39:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"48164cd8bf85271696bdf5efb2ba6afb3dafe96b2b7b4e9a371d857387748971","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-06-19T08:55:56Z","title_canon_sha256":"1c8176bcec8e2e4c59ecdc117c198c7d4bb960b68ef087eef061e001b1c802f7"},"schema_version":"1.0","source":{"id":"2406.13352","kind":"arxiv","version":3}},"canonical_sha256":"5f87c17f7a7ad7cc941de1399e778d0f051b0d4de487a14b8cf099a9dec360e8","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"5f87c17f7a7ad7cc941de1399e778d0f051b0d4de487a14b8cf099a9dec360e8","first_computed_at":"2026-07-05T09:39:56.114097Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:39:56.114097Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"wm4QeWhZ/2q5vlR+558mSiqD5g6APbZ/gESmjW8PXUJyELtEuctbHJKWwAoMWuOXftmbR+2TBhx6fAk9/zDOCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T09:39:56.114651Z","signed_message":"canonical_sha256_bytes"},"source_id":"2406.13352","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:2a07ab63b992e968affed581c64c154e4174ec74b2e314e97b1b662e1137c661","sha256:55078a5ed2b4a22be9aabfb23d7a59b229e9352f48e1a417b9e99006a7d750ca"],"state_sha256":"7f640d86948fce0592403b001ba78b7cdb68fb7e4fe80c8fa88c7498b71f23b7"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"7Q86XscITlsKK42l2W7RM0TLKD05Y2WUp5gQ1G4RU4wDRxxg+qwukPPpLIz1htEN+QEgrDvxaXiKyn8WCS5QAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T19:24:31.788099Z","bundle_sha256":"7c4745a6eff976fb6025efaaa656bd8c97a39fa433cdfbc3571ede057b52d91e"}}