{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3TJ4WG7R6PLANPSL6OAMGEDGXD","short_pith_number":"pith:3TJ4WG7R","schema_version":"1.0","canonical_sha256":"dcd3cb1bf1f3d606be4bf380c31066b8c016d30dc9f1cb8d771ecf20444a1a4e","source":{"kind":"arxiv","id":"2405.20778","version":2},"attestation_state":"computed","paper":{"title":"Improved Generation of Adversarial Examples Against Safety-aligned LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Hao Chen, Qizhang Li, Wangmeng Zuo, Yiwen Guo","submitted_at":"2024-05-28T06:10:12Z","abstract_excerpt":"Adversarial prompts generated using gradient-based methods exhibit outstanding performance in performing automatic jailbreak attacks against safety-aligned LLMs. Nevertheless, due to the discrete nature of texts, the input gradient of LLMs struggles to precisely reflect the magnitude of loss change that results from token replacements in the prompt, leading to limited attack success rates against safety-aligned LLMs, even in the white-box setting. In this paper, we explore a new perspective on this problem, suggesting that it can be alleviated by leveraging innovations inspired in transfer-bas"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.20778","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2024-05-28T06:10:12Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"d502c8ded6fc01d623995de21abecc3e2aab688dc80e7d5ef1a5db192a952545","abstract_canon_sha256":"101d1ddb5c974f88448c639388f656cced1fa816e7c128e71cc6caabda0379cf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:29:29.644734Z","signature_b64":"Gtq1AX036XRPN+joKl32uwZzx4i3bMVJnYaJ6XG4Md8HnnoaygSrbow+IuY0g7WqIY44Hi0458hzCbgz2+HkCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dcd3cb1bf1f3d606be4bf380c31066b8c016d30dc9f1cb8d771ecf20444a1a4e","last_reissued_at":"2026-07-05T09:29:29.644253Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:29:29.644253Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improved Generation of Adversarial Examples Against Safety-aligned LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Hao Chen, Qizhang Li, Wangmeng Zuo, Yiwen Guo","submitted_at":"2024-05-28T06:10:12Z","abstract_excerpt":"Adversarial prompts generated using gradient-based methods exhibit outstanding performance in performing automatic jailbreak attacks against safety-aligned LLMs. Nevertheless, due to the discrete nature of texts, the input gradient of LLMs struggles to precisely reflect the magnitude of loss change that results from token replacements in the prompt, leading to limited attack success rates against safety-aligned LLMs, even in the white-box setting. In this paper, we explore a new perspective on this problem, suggesting that it can be alleviated by leveraging innovations inspired in transfer-bas"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.20778","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.20778/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.20778","created_at":"2026-07-05T09:29:29.644307+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.20778v2","created_at":"2026-07-05T09:29:29.644307+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.20778","created_at":"2026-07-05T09:29:29.644307+00:00"},{"alias_kind":"pith_short_12","alias_value":"3TJ4WG7R6PLA","created_at":"2026-07-05T09:29:29.644307+00:00"},{"alias_kind":"pith_short_16","alias_value":"3TJ4WG7R6PLANPSL","created_at":"2026-07-05T09:29:29.644307+00:00"},{"alias_kind":"pith_short_8","alias_value":"3TJ4WG7R","created_at":"2026-07-05T09:29:29.644307+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.00391","citing_title":"The Resurgence of GCG Adversarial Attacks on Large Language Models","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3TJ4WG7R6PLANPSL6OAMGEDGXD","json":"https://pith.science/pith/3TJ4WG7R6PLANPSL6OAMGEDGXD.json","graph_json":"https://pith.science/api/pith-number/3TJ4WG7R6PLANPSL6OAMGEDGXD/graph.json","events_json":"https://pith.science/api/pith-number/3TJ4WG7R6PLANPSL6OAMGEDGXD/events.json","paper":"https://pith.science/paper/3TJ4WG7R"},"agent_actions":{"view_html":"https://pith.science/pith/3TJ4WG7R6PLANPSL6OAMGEDGXD","download_json":"https://pith.science/pith/3TJ4WG7R6PLANPSL6OAMGEDGXD.json","view_paper":"https://pith.science/paper/3TJ4WG7R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.20778&json=true","fetch_graph":"https://pith.science/api/pith-number/3TJ4WG7R6PLANPSL6OAMGEDGXD/graph.json","fetch_events":"https://pith.science/api/pith-number/3TJ4WG7R6PLANPSL6OAMGEDGXD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3TJ4WG7R6PLANPSL6OAMGEDGXD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3TJ4WG7R6PLANPSL6OAMGEDGXD/action/storage_attestation","attest_author":"https://pith.science/pith/3TJ4WG7R6PLANPSL6OAMGEDGXD/action/author_attestation","sign_citation":"https://pith.science/pith/3TJ4WG7R6PLANPSL6OAMGEDGXD/action/citation_signature","submit_replication":"https://pith.science/pith/3TJ4WG7R6PLANPSL6OAMGEDGXD/action/replication_record"}},"created_at":"2026-07-05T09:29:29.644307+00:00","updated_at":"2026-07-05T09:29:29.644307+00:00"}