{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SFG7YWVY6OBF2I3VDYNAFITSIV","short_pith_number":"pith:SFG7YWVY","schema_version":"1.0","canonical_sha256":"914dfc5ab8f3825d23751e1a02a272454fdadd45ca09d14a001cadf2043167f4","source":{"kind":"arxiv","id":"2504.11168","version":3},"attestation_state":"computed","paper":{"title":"Bypassing LLM Guardrails: An Empirical Analysis of Evasion Attacks against Prompt Injection and Jailbreak Detection Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CR","authors_text":"Lewis Birch, Neeraj Suri, Peter Garraghan, Stefan Trawicki, William Hackett","submitted_at":"2025-04-15T13:16:02Z","abstract_excerpt":"Large Language Models (LLMs) guardrail systems are designed to protect against prompt injection and jailbreak attacks. However, they remain vulnerable to evasion techniques. We demonstrate two approaches for bypassing LLM prompt injection and jailbreak detection systems via traditional character injection methods and algorithmic Adversarial Machine Learning (AML) evasion techniques. Through testing against six prominent protection systems, including Microsoft's Azure Prompt Shield and Meta's Prompt Guard, we show that both methods can be used to evade detection while maintaining adversarial ut"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.11168","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2025-04-15T13:16:02Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"e26c25e4dd0d9973636f8b25fca37b1e263309f1d0251838df5b16e736576a66","abstract_canon_sha256":"6b87bccf499cbf355bc5aaec239785226653971f3d9eda70500ce8c9e24c9cde"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:30.760353Z","signature_b64":"uzWJ+YJCwDQTiyDYihKl7WzuNkntnHcq37anEiNH7+cjMTl64tmNyeCPSp/DusUDo7m+AL0qgKbkP63BUEdeAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"914dfc5ab8f3825d23751e1a02a272454fdadd45ca09d14a001cadf2043167f4","last_reissued_at":"2026-07-05T11:36:30.759870Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:30.759870Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Bypassing LLM Guardrails: An Empirical Analysis of Evasion Attacks against Prompt Injection and Jailbreak Detection Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CR","authors_text":"Lewis Birch, Neeraj Suri, Peter Garraghan, Stefan Trawicki, William Hackett","submitted_at":"2025-04-15T13:16:02Z","abstract_excerpt":"Large Language Models (LLMs) guardrail systems are designed to protect against prompt injection and jailbreak attacks. However, they remain vulnerable to evasion techniques. We demonstrate two approaches for bypassing LLM prompt injection and jailbreak detection systems via traditional character injection methods and algorithmic Adversarial Machine Learning (AML) evasion techniques. Through testing against six prominent protection systems, including Microsoft's Azure Prompt Shield and Meta's Prompt Guard, we show that both methods can be used to evade detection while maintaining adversarial ut"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.11168","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.11168/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.11168","created_at":"2026-07-05T11:36:30.759930+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.11168v3","created_at":"2026-07-05T11:36:30.759930+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.11168","created_at":"2026-07-05T11:36:30.759930+00:00"},{"alias_kind":"pith_short_12","alias_value":"SFG7YWVY6OBF","created_at":"2026-07-05T11:36:30.759930+00:00"},{"alias_kind":"pith_short_16","alias_value":"SFG7YWVY6OBF2I3V","created_at":"2026-07-05T11:36:30.759930+00:00"},{"alias_kind":"pith_short_8","alias_value":"SFG7YWVY","created_at":"2026-07-05T11:36:30.759930+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22237","citing_title":"Investigating The Security of Modern AI and Cloud Infrastructure","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24598","citing_title":"Toward Self-Evolution-Ready Workflow Harnesses: A Reversible Migration Path and Convertibility Taxonomy for Expert LLM Pipelines","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.14517","citing_title":"From Shield to Target: Denial-of-Service Attacks on LLM-Based Agent Guardrails","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03136","citing_title":"PsychoPass: Geometric Profiling of Multi-Turn Adversarial LLM Conversations","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05946","citing_title":"Short paper: Models in the dark -- Rectification and erasure under GDPR in ML supply chains","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2504.14668","citing_title":"A Byzantine Fault Tolerance Approach towards AI Safety","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2510.09093","citing_title":"Exploiting Web Search Tools of AI Agents for Data Exfiltration","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SFG7YWVY6OBF2I3VDYNAFITSIV","json":"https://pith.science/pith/SFG7YWVY6OBF2I3VDYNAFITSIV.json","graph_json":"https://pith.science/api/pith-number/SFG7YWVY6OBF2I3VDYNAFITSIV/graph.json","events_json":"https://pith.science/api/pith-number/SFG7YWVY6OBF2I3VDYNAFITSIV/events.json","paper":"https://pith.science/paper/SFG7YWVY"},"agent_actions":{"view_html":"https://pith.science/pith/SFG7YWVY6OBF2I3VDYNAFITSIV","download_json":"https://pith.science/pith/SFG7YWVY6OBF2I3VDYNAFITSIV.json","view_paper":"https://pith.science/paper/SFG7YWVY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.11168&json=true","fetch_graph":"https://pith.science/api/pith-number/SFG7YWVY6OBF2I3VDYNAFITSIV/graph.json","fetch_events":"https://pith.science/api/pith-number/SFG7YWVY6OBF2I3VDYNAFITSIV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SFG7YWVY6OBF2I3VDYNAFITSIV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SFG7YWVY6OBF2I3VDYNAFITSIV/action/storage_attestation","attest_author":"https://pith.science/pith/SFG7YWVY6OBF2I3VDYNAFITSIV/action/author_attestation","sign_citation":"https://pith.science/pith/SFG7YWVY6OBF2I3VDYNAFITSIV/action/citation_signature","submit_replication":"https://pith.science/pith/SFG7YWVY6OBF2I3VDYNAFITSIV/action/replication_record"}},"created_at":"2026-07-05T11:36:30.759930+00:00","updated_at":"2026-07-05T11:36:30.759930+00:00"}