{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MBBCYRW6NROTE64AGNQ5HOUMQN","short_pith_number":"pith:MBBCYRW6","schema_version":"1.0","canonical_sha256":"60422c46de6c5d327b803361d3ba8c8362f30d975208a4fd361ae2aba26a5560","source":{"kind":"arxiv","id":"2402.08983","version":4},"attestation_state":"computed","paper":{"title":"SafeDecoding: Defending against Jailbreak Attacks via Safety-Aware Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CR","authors_text":"Bill Yuchen Lin, Fengqing Jiang, Jinyuan Jia, Luyao Niu, Radha Poovendran, Zhangchen Xu","submitted_at":"2024-02-14T06:54:31Z","abstract_excerpt":"As large language models (LLMs) become increasingly integrated into real-world applications such as code generation and chatbot assistance, extensive efforts have been made to align LLM behavior with human values, including safety. Jailbreak attacks, aiming to provoke unintended and unsafe behaviors from LLMs, remain a significant/leading LLM safety threat. In this paper, we aim to defend LLMs against jailbreak attacks by introducing SafeDecoding, a safety-aware decoding strategy for LLMs to generate helpful and harmless responses to user queries. Our insight in developing SafeDecoding is base"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.08983","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2024-02-14T06:54:31Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"0ce4061727074ac4fc22b1656cbec25cc064ba6b4f7bee34545182db619934f8","abstract_canon_sha256":"f858b4341ecae7901441d136ebda6c18cccce87ecc228cfaf524d3ebe8a7ba35"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:48:41.303163Z","signature_b64":"GQUjSWqCKyGIxsxX2feEXhDGRr+m2o+6vA55M5nY6S2co8Qu/FK/D7ie/FPbynSKUcDgHb5+mfCV4QDm9VeODw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"60422c46de6c5d327b803361d3ba8c8362f30d975208a4fd361ae2aba26a5560","last_reissued_at":"2026-07-05T08:48:41.302689Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:48:41.302689Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SafeDecoding: Defending against Jailbreak Attacks via Safety-Aware Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CR","authors_text":"Bill Yuchen Lin, Fengqing Jiang, Jinyuan Jia, Luyao Niu, Radha Poovendran, Zhangchen Xu","submitted_at":"2024-02-14T06:54:31Z","abstract_excerpt":"As large language models (LLMs) become increasingly integrated into real-world applications such as code generation and chatbot assistance, extensive efforts have been made to align LLM behavior with human values, including safety. Jailbreak attacks, aiming to provoke unintended and unsafe behaviors from LLMs, remain a significant/leading LLM safety threat. In this paper, we aim to defend LLMs against jailbreak attacks by introducing SafeDecoding, a safety-aware decoding strategy for LLMs to generate helpful and harmless responses to user queries. Our insight in developing SafeDecoding is base"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.08983","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.08983/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.08983","created_at":"2026-07-05T08:48:41.302756+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.08983v4","created_at":"2026-07-05T08:48:41.302756+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.08983","created_at":"2026-07-05T08:48:41.302756+00:00"},{"alias_kind":"pith_short_12","alias_value":"MBBCYRW6NROT","created_at":"2026-07-05T08:48:41.302756+00:00"},{"alias_kind":"pith_short_16","alias_value":"MBBCYRW6NROTE64A","created_at":"2026-07-05T08:48:41.302756+00:00"},{"alias_kind":"pith_short_8","alias_value":"MBBCYRW6","created_at":"2026-07-05T08:48:41.302756+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19755","citing_title":"SafeSpec: Fast and Safe LLM via Dynamic Reflective Sampling","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05609","citing_title":"SlotGCG: Exploiting the Positional Vulnerability in LLMs for Jailbreak Attacks","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03486","citing_title":"NeuroArmor: Safe-Variant-Guided Representation Consistency for Selective Re-Anchoring in Jailbreak Defense","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28153","citing_title":"Robust Harmful Features Under Jailbreak Attacks: Mechanistic Evidence from Attention Head Specialization in Large Language Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2508.04204","citing_title":"ReasoningGuard: Safeguarding Large Reasoning Models with Inference-time Safety Aha Moments","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11716","citing_title":"SafeSteer: A Decoding-level Defense Mechanism for Multimodal Large Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24082","citing_title":"Jailbreaking Frontier Foundation Models Through Intention Deception","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00123","citing_title":"Minimal, Local, Causal Explanations for Jailbreak Success in Large Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10326","citing_title":"Jailbreaking the Matrix: Nullspace Steering for Controlled Model Subversion","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07727","citing_title":"TrajGuard: Streaming Hidden-state Trajectory Detection for Decoding-time Jailbreak Defense","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MBBCYRW6NROTE64AGNQ5HOUMQN","json":"https://pith.science/pith/MBBCYRW6NROTE64AGNQ5HOUMQN.json","graph_json":"https://pith.science/api/pith-number/MBBCYRW6NROTE64AGNQ5HOUMQN/graph.json","events_json":"https://pith.science/api/pith-number/MBBCYRW6NROTE64AGNQ5HOUMQN/events.json","paper":"https://pith.science/paper/MBBCYRW6"},"agent_actions":{"view_html":"https://pith.science/pith/MBBCYRW6NROTE64AGNQ5HOUMQN","download_json":"https://pith.science/pith/MBBCYRW6NROTE64AGNQ5HOUMQN.json","view_paper":"https://pith.science/paper/MBBCYRW6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.08983&json=true","fetch_graph":"https://pith.science/api/pith-number/MBBCYRW6NROTE64AGNQ5HOUMQN/graph.json","fetch_events":"https://pith.science/api/pith-number/MBBCYRW6NROTE64AGNQ5HOUMQN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MBBCYRW6NROTE64AGNQ5HOUMQN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MBBCYRW6NROTE64AGNQ5HOUMQN/action/storage_attestation","attest_author":"https://pith.science/pith/MBBCYRW6NROTE64AGNQ5HOUMQN/action/author_attestation","sign_citation":"https://pith.science/pith/MBBCYRW6NROTE64AGNQ5HOUMQN/action/citation_signature","submit_replication":"https://pith.science/pith/MBBCYRW6NROTE64AGNQ5HOUMQN/action/replication_record"}},"created_at":"2026-07-05T08:48:41.302756+00:00","updated_at":"2026-07-05T08:48:41.302756+00:00"}