{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:C2Z7526C6PQIU67ROLGIV2DLUT","short_pith_number":"pith:C2Z7526C","schema_version":"1.0","canonical_sha256":"16b3feebc2f3e08a7bf172cc8ae86ba4cc8774616a6aad70fb03b7b014228b53","source":{"kind":"arxiv","id":"2410.09040","version":1},"attestation_state":"computed","paper":{"title":"AttnGCG: Enhancing Jailbreaking Attacks on LLMs with Attention Manipulation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bingchen Zhao, Cihang Xie, Haoqin Tu, Jieru Mei, Yisen Wang, Zijun Wang","submitted_at":"2024-10-11T17:55:09Z","abstract_excerpt":"This paper studies the vulnerabilities of transformer-based Large Language Models (LLMs) to jailbreaking attacks, focusing specifically on the optimization-based Greedy Coordinate Gradient (GCG) strategy. We first observe a positive correlation between the effectiveness of attacks and the internal behaviors of the models. For instance, attacks tend to be less effective when models pay more attention to system prompts designed to ensure LLM safety alignment. Building on this discovery, we introduce an enhanced method that manipulates models' attention scores to facilitate LLM jailbreaking, whic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.09040","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-11T17:55:09Z","cross_cats_sorted":[],"title_canon_sha256":"82f96297b9eef8e48bfa88d38df421caa74db0db8ac57e42810cba51c8356df7","abstract_canon_sha256":"c76d26bfb8b6bcd1977a40410e58af078a61bc3a455800de43d236f0f963fab4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:19:21.108951Z","signature_b64":"MliCMEoJQj5pF1woQKSnZJ/wI9z7mYJW7MHUzEu0Edy3IwC+MYhhVsTmGMk9BJfq+WR8unAP5xrwVFQckYYzCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"16b3feebc2f3e08a7bf172cc8ae86ba4cc8774616a6aad70fb03b7b014228b53","last_reissued_at":"2026-07-05T09:19:21.108508Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:19:21.108508Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AttnGCG: Enhancing Jailbreaking Attacks on LLMs with Attention Manipulation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bingchen Zhao, Cihang Xie, Haoqin Tu, Jieru Mei, Yisen Wang, Zijun Wang","submitted_at":"2024-10-11T17:55:09Z","abstract_excerpt":"This paper studies the vulnerabilities of transformer-based Large Language Models (LLMs) to jailbreaking attacks, focusing specifically on the optimization-based Greedy Coordinate Gradient (GCG) strategy. We first observe a positive correlation between the effectiveness of attacks and the internal behaviors of the models. For instance, attacks tend to be less effective when models pay more attention to system prompts designed to ensure LLM safety alignment. Building on this discovery, we introduce an enhanced method that manipulates models' attention scores to facilitate LLM jailbreaking, whic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.09040","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.09040/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.09040","created_at":"2026-07-05T09:19:21.108574+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.09040v1","created_at":"2026-07-05T09:19:21.108574+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.09040","created_at":"2026-07-05T09:19:21.108574+00:00"},{"alias_kind":"pith_short_12","alias_value":"C2Z7526C6PQI","created_at":"2026-07-05T09:19:21.108574+00:00"},{"alias_kind":"pith_short_16","alias_value":"C2Z7526C6PQIU67R","created_at":"2026-07-05T09:19:21.108574+00:00"},{"alias_kind":"pith_short_8","alias_value":"C2Z7526C","created_at":"2026-07-05T09:19:21.108574+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05609","citing_title":"SlotGCG: Exploiting the Positional Vulnerability in LLMs for Jailbreak Attacks","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2508.04204","citing_title":"ReasoningGuard: Safeguarding Large Reasoning Models with Inference-time Safety Aha Moments","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02946","citing_title":"RouteHijack: Routing-Aware Attack on Mixture-of-Experts LLMs","ref_index":68,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C2Z7526C6PQIU67ROLGIV2DLUT","json":"https://pith.science/pith/C2Z7526C6PQIU67ROLGIV2DLUT.json","graph_json":"https://pith.science/api/pith-number/C2Z7526C6PQIU67ROLGIV2DLUT/graph.json","events_json":"https://pith.science/api/pith-number/C2Z7526C6PQIU67ROLGIV2DLUT/events.json","paper":"https://pith.science/paper/C2Z7526C"},"agent_actions":{"view_html":"https://pith.science/pith/C2Z7526C6PQIU67ROLGIV2DLUT","download_json":"https://pith.science/pith/C2Z7526C6PQIU67ROLGIV2DLUT.json","view_paper":"https://pith.science/paper/C2Z7526C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.09040&json=true","fetch_graph":"https://pith.science/api/pith-number/C2Z7526C6PQIU67ROLGIV2DLUT/graph.json","fetch_events":"https://pith.science/api/pith-number/C2Z7526C6PQIU67ROLGIV2DLUT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C2Z7526C6PQIU67ROLGIV2DLUT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C2Z7526C6PQIU67ROLGIV2DLUT/action/storage_attestation","attest_author":"https://pith.science/pith/C2Z7526C6PQIU67ROLGIV2DLUT/action/author_attestation","sign_citation":"https://pith.science/pith/C2Z7526C6PQIU67ROLGIV2DLUT/action/citation_signature","submit_replication":"https://pith.science/pith/C2Z7526C6PQIU67ROLGIV2DLUT/action/replication_record"}},"created_at":"2026-07-05T09:19:21.108574+00:00","updated_at":"2026-07-05T09:19:21.108574+00:00"}