{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DB6OCLO76QFE6TYWDMJ2JA3XVJ","short_pith_number":"pith:DB6OCLO7","schema_version":"1.0","canonical_sha256":"187ce12ddff40a4f4f161b13a48377aa7cac299eda281361df3bb79b8313f729","source":{"kind":"arxiv","id":"2402.16192","version":2},"attestation_state":"computed","paper":{"title":"Defending Large Language Models against Jailbreak Attacks via Semantic Smoothing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexander Robey, Bairu Hou, Eric Wong, George J. Pappas, Hamed Hassani, Jiabao Ji, Shiyu Chang, Yang Zhang","submitted_at":"2024-02-25T20:36:03Z","abstract_excerpt":"Aligned large language models (LLMs) are vulnerable to jailbreaking attacks, which bypass the safeguards of targeted LLMs and fool them into generating objectionable content. While initial defenses show promise against token-based threat models, there do not exist defenses that provide robustness against semantic attacks and avoid unfavorable trade-offs between robustness and nominal performance. To meet this need, we propose SEMANTICSMOOTH, a smoothing-based defense that aggregates the predictions of multiple semantically transformed copies of a given input prompt. Experimental results demons"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.16192","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-25T20:36:03Z","cross_cats_sorted":[],"title_canon_sha256":"3dd6dff03031a89f165919f91829fcbd3e0f601b184e98b156d9cbf16e8394b2","abstract_canon_sha256":"6a9075f11c6ab213de02bb0acd6f10d988a02d7377c2277cdbcf2fd1f2489a66"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:50:23.806259Z","signature_b64":"+VglMAG45GdgGthqeFgtTYtAu4O72c1JwSoK2kE47xs9nv3szMST5IGdvNmpW5LFaBWSSRTa6cC5+Bky/FL3Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"187ce12ddff40a4f4f161b13a48377aa7cac299eda281361df3bb79b8313f729","last_reissued_at":"2026-07-05T07:50:23.805749Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:50:23.805749Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Defending Large Language Models against Jailbreak Attacks via Semantic Smoothing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexander Robey, Bairu Hou, Eric Wong, George J. Pappas, Hamed Hassani, Jiabao Ji, Shiyu Chang, Yang Zhang","submitted_at":"2024-02-25T20:36:03Z","abstract_excerpt":"Aligned large language models (LLMs) are vulnerable to jailbreaking attacks, which bypass the safeguards of targeted LLMs and fool them into generating objectionable content. While initial defenses show promise against token-based threat models, there do not exist defenses that provide robustness against semantic attacks and avoid unfavorable trade-offs between robustness and nominal performance. To meet this need, we propose SEMANTICSMOOTH, a smoothing-based defense that aggregates the predictions of multiple semantically transformed copies of a given input prompt. Experimental results demons"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.16192","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.16192/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.16192","created_at":"2026-07-05T07:50:23.805810+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.16192v2","created_at":"2026-07-05T07:50:23.805810+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.16192","created_at":"2026-07-05T07:50:23.805810+00:00"},{"alias_kind":"pith_short_12","alias_value":"DB6OCLO76QFE","created_at":"2026-07-05T07:50:23.805810+00:00"},{"alias_kind":"pith_short_16","alias_value":"DB6OCLO76QFE6TYW","created_at":"2026-07-05T07:50:23.805810+00:00"},{"alias_kind":"pith_short_8","alias_value":"DB6OCLO7","created_at":"2026-07-05T07:50:23.805810+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05609","citing_title":"SlotGCG: Exploiting the Positional Vulnerability in LLMs for Jailbreak Attacks","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24552","citing_title":"Ellipsoid Control: A White-list Jailbreak Defense via Benign Latent Modeling","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05206","citing_title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2508.04204","citing_title":"ReasoningGuard: Safeguarding Large Reasoning Models with Inference-time Safety Aha Moments","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2404.01318","citing_title":"JailbreakBench: An Open Robustness Benchmark for Jailbreaking Large Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2310.08419","citing_title":"Jailbreaking Black Box Large Language Models in Twenty Queries","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18756","citing_title":"Towards Understanding the Robustness of Sparse Autoencoders","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DB6OCLO76QFE6TYWDMJ2JA3XVJ","json":"https://pith.science/pith/DB6OCLO76QFE6TYWDMJ2JA3XVJ.json","graph_json":"https://pith.science/api/pith-number/DB6OCLO76QFE6TYWDMJ2JA3XVJ/graph.json","events_json":"https://pith.science/api/pith-number/DB6OCLO76QFE6TYWDMJ2JA3XVJ/events.json","paper":"https://pith.science/paper/DB6OCLO7"},"agent_actions":{"view_html":"https://pith.science/pith/DB6OCLO76QFE6TYWDMJ2JA3XVJ","download_json":"https://pith.science/pith/DB6OCLO76QFE6TYWDMJ2JA3XVJ.json","view_paper":"https://pith.science/paper/DB6OCLO7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.16192&json=true","fetch_graph":"https://pith.science/api/pith-number/DB6OCLO76QFE6TYWDMJ2JA3XVJ/graph.json","fetch_events":"https://pith.science/api/pith-number/DB6OCLO76QFE6TYWDMJ2JA3XVJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DB6OCLO76QFE6TYWDMJ2JA3XVJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DB6OCLO76QFE6TYWDMJ2JA3XVJ/action/storage_attestation","attest_author":"https://pith.science/pith/DB6OCLO76QFE6TYWDMJ2JA3XVJ/action/author_attestation","sign_citation":"https://pith.science/pith/DB6OCLO76QFE6TYWDMJ2JA3XVJ/action/citation_signature","submit_replication":"https://pith.science/pith/DB6OCLO76QFE6TYWDMJ2JA3XVJ/action/replication_record"}},"created_at":"2026-07-05T07:50:23.805810+00:00","updated_at":"2026-07-05T07:50:23.805810+00:00"}