{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TLPE5OUE3W33RVKIVOWNDER4VQ","short_pith_number":"pith:TLPE5OUE","schema_version":"1.0","canonical_sha256":"9ade4eba84ddb7b8d548abacd1923cac394fb4fae77acf2ece50131e273c08a3","source":{"kind":"arxiv","id":"2309.02705","version":4},"attestation_state":"computed","paper":{"title":"Certifying LLM Safety against Adversarial Prompting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aaron Jiaxun Li, Aounon Kumar, Chirag Agarwal, Himabindu Lakkaraju, Soheil Feizi, Suraj Srinivas","submitted_at":"2023-09-06T04:37:20Z","abstract_excerpt":"Large language models (LLMs) are vulnerable to adversarial attacks that add malicious tokens to an input prompt to bypass the safety guardrails of an LLM and cause it to produce harmful content. In this work, we introduce erase-and-check, the first framework for defending against adversarial prompts with certifiable safety guarantees. Given a prompt, our procedure erases tokens individually and inspects the resulting subsequences using a safety filter. Our safety certificate guarantees that harmful prompts are not mislabeled as safe due to an adversarial attack up to a certain size. We impleme"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.02705","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-06T04:37:20Z","cross_cats_sorted":["cs.AI","cs.CR","cs.LG"],"title_canon_sha256":"59d548770a84f2988a32e2d844b33a9ab5327f91331c059a1ae03a99abaeb00b","abstract_canon_sha256":"1e5dba70a485a250de5f413cf810a7be352a293404dd5f53bfb739202622e5e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:40.491255Z","signature_b64":"S5gbtiA/gOEIcuK3FePvXmYVytmY3HdJ8aCZu1sOTauNGQZwd6+YSjqB24KRnlm2FBGZMamUkdG/RTdSbimGDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9ade4eba84ddb7b8d548abacd1923cac394fb4fae77acf2ece50131e273c08a3","last_reissued_at":"2026-07-05T10:09:40.490839Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:40.490839Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Certifying LLM Safety against Adversarial Prompting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aaron Jiaxun Li, Aounon Kumar, Chirag Agarwal, Himabindu Lakkaraju, Soheil Feizi, Suraj Srinivas","submitted_at":"2023-09-06T04:37:20Z","abstract_excerpt":"Large language models (LLMs) are vulnerable to adversarial attacks that add malicious tokens to an input prompt to bypass the safety guardrails of an LLM and cause it to produce harmful content. In this work, we introduce erase-and-check, the first framework for defending against adversarial prompts with certifiable safety guarantees. Given a prompt, our procedure erases tokens individually and inspects the resulting subsequences using a safety filter. Our safety certificate guarantees that harmful prompts are not mislabeled as safe due to an adversarial attack up to a certain size. We impleme"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.02705","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.02705/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.02705","created_at":"2026-07-05T10:09:40.490894+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.02705v4","created_at":"2026-07-05T10:09:40.490894+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.02705","created_at":"2026-07-05T10:09:40.490894+00:00"},{"alias_kind":"pith_short_12","alias_value":"TLPE5OUE3W33","created_at":"2026-07-05T10:09:40.490894+00:00"},{"alias_kind":"pith_short_16","alias_value":"TLPE5OUE3W33RVKI","created_at":"2026-07-05T10:09:40.490894+00:00"},{"alias_kind":"pith_short_8","alias_value":"TLPE5OUE","created_at":"2026-07-05T10:09:40.490894+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07461","citing_title":"Mitigating Taint-Style Vulnerabilities in MCP Servers via Security-Aware Tool Descriptions","ref_index":23,"is_internal_anchor":true},{"citing_arxiv_id":"2606.05609","citing_title":"SlotGCG: Exploiting the Positional Vulnerability in LLMs for Jailbreak Attacks","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04141","citing_title":"Caught in the Act(ivation): Toward Pre-Output and Multi-Turn Detection of Credential Exfiltration by LLM Agents","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2409.10102","citing_title":"Trustworthiness in Retrieval-Augmented Generation Systems: A Survey","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05206","citing_title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16120","citing_title":"LLM-Powered AI Agent Systems and Their Applications in Industry","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21674","citing_title":"Adversarial Reframing: A Framework for Targeted Generation in Language Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19485","citing_title":"Attention-Guided Reward for Reinforcement Learning-based Jailbreak against Large Reasoning Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2510.09689","citing_title":"When Search Goes Wrong: Red-Teaming Web-Augmented Large Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2512.05439","citing_title":"BEAVER: An Efficient Deterministic LLM Verifier","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2404.01318","citing_title":"JailbreakBench: An Open Robustness Benchmark for Jailbreaking Large Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2310.03684","citing_title":"SmoothLLM: Defending Large Language Models Against Jailbreaking Attacks","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03976","citing_title":"Quantifying Trust: Financial Risk Management for Trustworthy AI Agents","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10611","citing_title":"Re-Triggering Safeguards within LLMs for Jailbreak Detection","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10516","citing_title":"Consistency as a Testable Property: Statistical Methods to Evaluate AI Agent Reliability","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24162","citing_title":"Defusing the Trigger: Plug-and-Play Defense for Backdoored LLMs via Tail-Risk Intrinsic Geometric Smoothing","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24700","citing_title":"Green Shielding: A User-Centric Approach Towards Trustworthy AI","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06605","citing_title":"How Many Iterations to Jailbreak? Dynamic Budget Allocation for Multi-Turn LLM Evaluation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00236","citing_title":"Attention Is Where You Attack","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22089","citing_title":"Ethics Testing: Proactive Identification of Generative AI System Harms","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09747","citing_title":"ADAM: A Systematic Data Extraction Attack on Agent Memory via Adaptive Querying","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18756","citing_title":"Towards Understanding the Robustness of Sparse Autoencoders","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TLPE5OUE3W33RVKIVOWNDER4VQ","json":"https://pith.science/pith/TLPE5OUE3W33RVKIVOWNDER4VQ.json","graph_json":"https://pith.science/api/pith-number/TLPE5OUE3W33RVKIVOWNDER4VQ/graph.json","events_json":"https://pith.science/api/pith-number/TLPE5OUE3W33RVKIVOWNDER4VQ/events.json","paper":"https://pith.science/paper/TLPE5OUE"},"agent_actions":{"view_html":"https://pith.science/pith/TLPE5OUE3W33RVKIVOWNDER4VQ","download_json":"https://pith.science/pith/TLPE5OUE3W33RVKIVOWNDER4VQ.json","view_paper":"https://pith.science/paper/TLPE5OUE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.02705&json=true","fetch_graph":"https://pith.science/api/pith-number/TLPE5OUE3W33RVKIVOWNDER4VQ/graph.json","fetch_events":"https://pith.science/api/pith-number/TLPE5OUE3W33RVKIVOWNDER4VQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TLPE5OUE3W33RVKIVOWNDER4VQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TLPE5OUE3W33RVKIVOWNDER4VQ/action/storage_attestation","attest_author":"https://pith.science/pith/TLPE5OUE3W33RVKIVOWNDER4VQ/action/author_attestation","sign_citation":"https://pith.science/pith/TLPE5OUE3W33RVKIVOWNDER4VQ/action/citation_signature","submit_replication":"https://pith.science/pith/TLPE5OUE3W33RVKIVOWNDER4VQ/action/replication_record"}},"created_at":"2026-07-05T10:09:40.490894+00:00","updated_at":"2026-07-05T10:09:40.490894+00:00"}