{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:O4U6KDNOXZY7KXBFNFY37JPBUV","short_pith_number":"pith:O4U6KDNO","schema_version":"1.0","canonical_sha256":"7729e50daebe71f55c256971bfa5e1a56433763b73095d18f2d85930a60a0b6c","source":{"kind":"arxiv","id":"2411.07494","version":1},"attestation_state":"computed","paper":{"title":"Rapid Response: Mitigating LLM Jailbreaks with a Few Examples","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alwin Peng, Ethan Perez, Henry Sleight, Julian Michael, Mrinank Sharma","submitted_at":"2024-11-12T02:44:49Z","abstract_excerpt":"As large language models (LLMs) grow more powerful, ensuring their safety against misuse becomes crucial. While researchers have focused on developing robust defenses, no method has yet achieved complete invulnerability to attacks. We propose an alternative approach: instead of seeking perfect adversarial robustness, we develop rapid response techniques to look to block whole classes of jailbreaks after observing only a handful of attacks. To study this setting, we develop RapidResponseBench, a benchmark that measures a defense's robustness against various jailbreak strategies after adapting t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.07494","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-12T02:44:49Z","cross_cats_sorted":[],"title_canon_sha256":"99dac75921c0ccfeb9fafa6d937e19b4902529c2b42a3e3e419893494dd5d9d1","abstract_canon_sha256":"ece1ddab9c27a92e1b4b8e09a3459f55d53e878331cac64c50dcce246625f2c9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:34:24.652734Z","signature_b64":"HyxHPTfLGoWgY02eUka2IbJev7KXgjN6/P4v/zpmdcwKdIrQUL8K7I8Gb/mX4vyFU3lYVefJrXEl5SYuduBQAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7729e50daebe71f55c256971bfa5e1a56433763b73095d18f2d85930a60a0b6c","last_reissued_at":"2026-07-05T09:34:24.652158Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:34:24.652158Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Rapid Response: Mitigating LLM Jailbreaks with a Few Examples","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alwin Peng, Ethan Perez, Henry Sleight, Julian Michael, Mrinank Sharma","submitted_at":"2024-11-12T02:44:49Z","abstract_excerpt":"As large language models (LLMs) grow more powerful, ensuring their safety against misuse becomes crucial. While researchers have focused on developing robust defenses, no method has yet achieved complete invulnerability to attacks. We propose an alternative approach: instead of seeking perfect adversarial robustness, we develop rapid response techniques to look to block whole classes of jailbreaks after observing only a handful of attacks. To study this setting, we develop RapidResponseBench, a benchmark that measures a defense's robustness against various jailbreak strategies after adapting t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.07494","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.07494/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.07494","created_at":"2026-07-05T09:34:24.652220+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.07494v1","created_at":"2026-07-05T09:34:24.652220+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.07494","created_at":"2026-07-05T09:34:24.652220+00:00"},{"alias_kind":"pith_short_12","alias_value":"O4U6KDNOXZY7","created_at":"2026-07-05T09:34:24.652220+00:00"},{"alias_kind":"pith_short_16","alias_value":"O4U6KDNOXZY7KXBF","created_at":"2026-07-05T09:34:24.652220+00:00"},{"alias_kind":"pith_short_8","alias_value":"O4U6KDNO","created_at":"2026-07-05T09:34:24.652220+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.28467","citing_title":"Mitigating Adaptive Attacks against Reasoning Models with Activation Consistency Training","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2506.00166","citing_title":"Disentangled Safety Adapters Enable Efficient Guardrails and Flexible Inference-Time Alignment","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2506.17299","citing_title":"Toward Principled LLM Safety Testing: Solving the Jailbreak Oracle Problem","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O4U6KDNOXZY7KXBFNFY37JPBUV","json":"https://pith.science/pith/O4U6KDNOXZY7KXBFNFY37JPBUV.json","graph_json":"https://pith.science/api/pith-number/O4U6KDNOXZY7KXBFNFY37JPBUV/graph.json","events_json":"https://pith.science/api/pith-number/O4U6KDNOXZY7KXBFNFY37JPBUV/events.json","paper":"https://pith.science/paper/O4U6KDNO"},"agent_actions":{"view_html":"https://pith.science/pith/O4U6KDNOXZY7KXBFNFY37JPBUV","download_json":"https://pith.science/pith/O4U6KDNOXZY7KXBFNFY37JPBUV.json","view_paper":"https://pith.science/paper/O4U6KDNO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.07494&json=true","fetch_graph":"https://pith.science/api/pith-number/O4U6KDNOXZY7KXBFNFY37JPBUV/graph.json","fetch_events":"https://pith.science/api/pith-number/O4U6KDNOXZY7KXBFNFY37JPBUV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O4U6KDNOXZY7KXBFNFY37JPBUV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O4U6KDNOXZY7KXBFNFY37JPBUV/action/storage_attestation","attest_author":"https://pith.science/pith/O4U6KDNOXZY7KXBFNFY37JPBUV/action/author_attestation","sign_citation":"https://pith.science/pith/O4U6KDNOXZY7KXBFNFY37JPBUV/action/citation_signature","submit_replication":"https://pith.science/pith/O4U6KDNOXZY7KXBFNFY37JPBUV/action/replication_record"}},"created_at":"2026-07-05T09:34:24.652220+00:00","updated_at":"2026-07-05T09:34:24.652220+00:00"}