{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EBOIS6IUYU7PPI7ZXGRWLTKUOB","short_pith_number":"pith:EBOIS6IU","schema_version":"1.0","canonical_sha256":"205c897914c53ef7a3f9b9a365cd547057326b40a7be0f34e3d49d929fad1b2c","source":{"kind":"arxiv","id":"2505.15753","version":1},"attestation_state":"computed","paper":{"title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CR","authors_text":"Ang Li, Taiye Chen, Yisen Wang, Zeming Wei","submitted_at":"2025-05-21T16:58:14Z","abstract_excerpt":"Large Language Models (LLMs) are known to be vulnerable to jailbreaking attacks, wherein adversaries exploit carefully engineered prompts to induce harmful or unethical responses. Such threats have raised critical concerns about the safety and reliability of LLMs in real-world deployment. While existing defense mechanisms partially mitigate such risks, subsequent advancements in adversarial techniques have enabled novel jailbreaking methods to circumvent these protections, exposing the limitations of static defense frameworks. In this work, we explore defending against evolving jailbreaking th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.15753","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2025-05-21T16:58:14Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"34fbcc2178fd39a46350de3decf946655490e746e60859dff20b91282d4d9293","abstract_canon_sha256":"3431c34a8a1a63d7dcd1ab246f7ece65f6dfd7e64261a181ba041cc38433af15"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:57.453675Z","signature_b64":"KJJf2fL/SsE0cDVepI4C/8JdLOgKr13F6lmr2eAkZ5z147k92Jnet0exEZGuCUYFE2GmeDkjUGq8DmyIJgdYDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"205c897914c53ef7a3f9b9a365cd547057326b40a7be0f34e3d49d929fad1b2c","last_reissued_at":"2026-07-05T11:06:57.453160Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:57.453160Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scalable Defense against In-the-wild Jailbreaking Attacks with Safety Context Retrieval","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CR","authors_text":"Ang Li, Taiye Chen, Yisen Wang, Zeming Wei","submitted_at":"2025-05-21T16:58:14Z","abstract_excerpt":"Large Language Models (LLMs) are known to be vulnerable to jailbreaking attacks, wherein adversaries exploit carefully engineered prompts to induce harmful or unethical responses. Such threats have raised critical concerns about the safety and reliability of LLMs in real-world deployment. While existing defense mechanisms partially mitigate such risks, subsequent advancements in adversarial techniques have enabled novel jailbreaking methods to circumvent these protections, exposing the limitations of static defense frameworks. In this work, we explore defending against evolving jailbreaking th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.15753","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.15753/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.15753","created_at":"2026-07-05T11:06:57.453218+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.15753v1","created_at":"2026-07-05T11:06:57.453218+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.15753","created_at":"2026-07-05T11:06:57.453218+00:00"},{"alias_kind":"pith_short_12","alias_value":"EBOIS6IUYU7P","created_at":"2026-07-05T11:06:57.453218+00:00"},{"alias_kind":"pith_short_16","alias_value":"EBOIS6IUYU7PPI7Z","created_at":"2026-07-05T11:06:57.453218+00:00"},{"alias_kind":"pith_short_8","alias_value":"EBOIS6IU","created_at":"2026-07-05T11:06:57.453218+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.01770","citing_title":"ReGA: Model-Based Safeguard for LLMs via Representation-Guided Abstraction","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12069","citing_title":"Rethinking Jailbreak Detection of Large Vision Language Models with Representational Contrastive Scoring","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EBOIS6IUYU7PPI7ZXGRWLTKUOB","json":"https://pith.science/pith/EBOIS6IUYU7PPI7ZXGRWLTKUOB.json","graph_json":"https://pith.science/api/pith-number/EBOIS6IUYU7PPI7ZXGRWLTKUOB/graph.json","events_json":"https://pith.science/api/pith-number/EBOIS6IUYU7PPI7ZXGRWLTKUOB/events.json","paper":"https://pith.science/paper/EBOIS6IU"},"agent_actions":{"view_html":"https://pith.science/pith/EBOIS6IUYU7PPI7ZXGRWLTKUOB","download_json":"https://pith.science/pith/EBOIS6IUYU7PPI7ZXGRWLTKUOB.json","view_paper":"https://pith.science/paper/EBOIS6IU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.15753&json=true","fetch_graph":"https://pith.science/api/pith-number/EBOIS6IUYU7PPI7ZXGRWLTKUOB/graph.json","fetch_events":"https://pith.science/api/pith-number/EBOIS6IUYU7PPI7ZXGRWLTKUOB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EBOIS6IUYU7PPI7ZXGRWLTKUOB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EBOIS6IUYU7PPI7ZXGRWLTKUOB/action/storage_attestation","attest_author":"https://pith.science/pith/EBOIS6IUYU7PPI7ZXGRWLTKUOB/action/author_attestation","sign_citation":"https://pith.science/pith/EBOIS6IUYU7PPI7ZXGRWLTKUOB/action/citation_signature","submit_replication":"https://pith.science/pith/EBOIS6IUYU7PPI7ZXGRWLTKUOB/action/replication_record"}},"created_at":"2026-07-05T11:06:57.453218+00:00","updated_at":"2026-07-05T11:06:57.453218+00:00"}