{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MVG2FPZMPZYZQBJCVOYB6EDEJU","short_pith_number":"pith:MVG2FPZM","schema_version":"1.0","canonical_sha256":"654da2bf2c7e71980522abb01f10644d2beb2ec993ea41766b86deff82e0f125","source":{"kind":"arxiv","id":"2507.22160","version":1},"attestation_state":"computed","paper":{"title":"Strategic Deflection: Defending LLMs from Logit Manipulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CR","authors_text":"Amal El Fallah Seghrouchni, Faissal Sehbaoui, Jihad Rbaiti, Yassine Rachidy, Youssef Hmamouche","submitted_at":"2025-07-29T18:46:56Z","abstract_excerpt":"With the growing adoption of Large Language Models (LLMs) in critical areas, ensuring their security against jailbreaking attacks is paramount. While traditional defenses primarily rely on refusing malicious prompts, recent logit-level attacks have demonstrated the ability to bypass these safeguards by directly manipulating the token-selection process during generation. We introduce Strategic Deflection (SDeflection), a defense that redefines the LLM's response to such advanced attacks. Instead of outright refusal, the model produces an answer that is semantically adjacent to the user's reques"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.22160","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2025-07-29T18:46:56Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"957c94b0d91ff3b29029071dea07ce8dc57b772cf237a49bd5eed8f7123667b5","abstract_canon_sha256":"c762df760924a62abb303cf86a998d05fb0e5cd50409a925a30f2513b51e1023"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:45:17.331093Z","signature_b64":"rrBvlbj9CnAN04patmWByaWhKekpIa/KB/BpODA0I72VRup+z703Gky0YQNz+YhJfbVdH2RWNxy549GNacwECw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"654da2bf2c7e71980522abb01f10644d2beb2ec993ea41766b86deff82e0f125","last_reissued_at":"2026-07-05T11:45:17.330593Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:45:17.330593Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Strategic Deflection: Defending LLMs from Logit Manipulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CR","authors_text":"Amal El Fallah Seghrouchni, Faissal Sehbaoui, Jihad Rbaiti, Yassine Rachidy, Youssef Hmamouche","submitted_at":"2025-07-29T18:46:56Z","abstract_excerpt":"With the growing adoption of Large Language Models (LLMs) in critical areas, ensuring their security against jailbreaking attacks is paramount. While traditional defenses primarily rely on refusing malicious prompts, recent logit-level attacks have demonstrated the ability to bypass these safeguards by directly manipulating the token-selection process during generation. We introduce Strategic Deflection (SDeflection), a defense that redefines the LLM's response to such advanced attacks. Instead of outright refusal, the model produces an answer that is semantically adjacent to the user's reques"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.22160","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.22160/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.22160","created_at":"2026-07-05T11:45:17.330650+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.22160v1","created_at":"2026-07-05T11:45:17.330650+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.22160","created_at":"2026-07-05T11:45:17.330650+00:00"},{"alias_kind":"pith_short_12","alias_value":"MVG2FPZMPZYZ","created_at":"2026-07-05T11:45:17.330650+00:00"},{"alias_kind":"pith_short_16","alias_value":"MVG2FPZMPZYZQBJC","created_at":"2026-07-05T11:45:17.330650+00:00"},{"alias_kind":"pith_short_8","alias_value":"MVG2FPZM","created_at":"2026-07-05T11:45:17.330650+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2512.20806","citing_title":"Safety Alignment of LMs via Non-cooperative Games","ref_index":43,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MVG2FPZMPZYZQBJCVOYB6EDEJU","json":"https://pith.science/pith/MVG2FPZMPZYZQBJCVOYB6EDEJU.json","graph_json":"https://pith.science/api/pith-number/MVG2FPZMPZYZQBJCVOYB6EDEJU/graph.json","events_json":"https://pith.science/api/pith-number/MVG2FPZMPZYZQBJCVOYB6EDEJU/events.json","paper":"https://pith.science/paper/MVG2FPZM"},"agent_actions":{"view_html":"https://pith.science/pith/MVG2FPZMPZYZQBJCVOYB6EDEJU","download_json":"https://pith.science/pith/MVG2FPZMPZYZQBJCVOYB6EDEJU.json","view_paper":"https://pith.science/paper/MVG2FPZM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.22160&json=true","fetch_graph":"https://pith.science/api/pith-number/MVG2FPZMPZYZQBJCVOYB6EDEJU/graph.json","fetch_events":"https://pith.science/api/pith-number/MVG2FPZMPZYZQBJCVOYB6EDEJU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MVG2FPZMPZYZQBJCVOYB6EDEJU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MVG2FPZMPZYZQBJCVOYB6EDEJU/action/storage_attestation","attest_author":"https://pith.science/pith/MVG2FPZMPZYZQBJCVOYB6EDEJU/action/author_attestation","sign_citation":"https://pith.science/pith/MVG2FPZMPZYZQBJCVOYB6EDEJU/action/citation_signature","submit_replication":"https://pith.science/pith/MVG2FPZMPZYZQBJCVOYB6EDEJU/action/replication_record"}},"created_at":"2026-07-05T11:45:17.330650+00:00","updated_at":"2026-07-05T11:45:17.330650+00:00"}