{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2EZUDFV2AZEKTYO2DVRIFNKKDL","short_pith_number":"pith:2EZUDFV2","schema_version":"1.0","canonical_sha256":"d1334196ba0648a9e1da1d6282b54a1af1c4a41767fe7926312d90ed5ce3fde8","source":{"kind":"arxiv","id":"2412.08615","version":2},"attestation_state":"computed","paper":{"title":"Exploiting the Index Gradients for Optimization-Based Jailbreaking on Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Haoyu Xu, Jiahui Li, Xing Wang, Yongchang Hao, Yu Hong","submitted_at":"2024-12-11T18:37:56Z","abstract_excerpt":"Despite the advancements in training Large Language Models (LLMs) with alignment techniques to enhance the safety of generated content, these models remain susceptible to jailbreak, an adversarial attack method that exposes security vulnerabilities in LLMs. Notably, the Greedy Coordinate Gradient (GCG) method has demonstrated the ability to automatically generate adversarial suffixes that jailbreak state-of-the-art LLMs. However, the optimization process involved in GCG is highly time-consuming, rendering the jailbreaking pipeline inefficient. In this paper, we investigate the process of GCG a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.08615","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-12-11T18:37:56Z","cross_cats_sorted":[],"title_canon_sha256":"89dd9eac89eda823b4947ed191c5dbdad422ed74d279eb525323e37b6045201a","abstract_canon_sha256":"ebf92eb5d19db7d6b2181f4f4b0298610eae6d1a365e2e0a7d003578996fa714"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:49:14.134691Z","signature_b64":"b9/5JL5MnT2BK620VpOjfLvFOFCjkAuc/6EUu6M9+vi4cAXKoP2UOCnq3+Vkqmaw48MWIwD/KZfTfPE4DsiGDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d1334196ba0648a9e1da1d6282b54a1af1c4a41767fe7926312d90ed5ce3fde8","last_reissued_at":"2026-07-05T09:49:14.134212Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:49:14.134212Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploiting the Index Gradients for Optimization-Based Jailbreaking on Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Haoyu Xu, Jiahui Li, Xing Wang, Yongchang Hao, Yu Hong","submitted_at":"2024-12-11T18:37:56Z","abstract_excerpt":"Despite the advancements in training Large Language Models (LLMs) with alignment techniques to enhance the safety of generated content, these models remain susceptible to jailbreak, an adversarial attack method that exposes security vulnerabilities in LLMs. Notably, the Greedy Coordinate Gradient (GCG) method has demonstrated the ability to automatically generate adversarial suffixes that jailbreak state-of-the-art LLMs. However, the optimization process involved in GCG is highly time-consuming, rendering the jailbreaking pipeline inefficient. In this paper, we investigate the process of GCG a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.08615","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.08615/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.08615","created_at":"2026-07-05T09:49:14.134271+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.08615v2","created_at":"2026-07-05T09:49:14.134271+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.08615","created_at":"2026-07-05T09:49:14.134271+00:00"},{"alias_kind":"pith_short_12","alias_value":"2EZUDFV2AZEK","created_at":"2026-07-05T09:49:14.134271+00:00"},{"alias_kind":"pith_short_16","alias_value":"2EZUDFV2AZEKTYO2","created_at":"2026-07-05T09:49:14.134271+00:00"},{"alias_kind":"pith_short_8","alias_value":"2EZUDFV2","created_at":"2026-07-05T09:49:14.134271+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05609","citing_title":"SlotGCG: Exploiting the Positional Vulnerability in LLMs for Jailbreak Attacks","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00150","citing_title":"Persona Attack: Incremental Memory Injection Jailbreak Attack against Large Language Models","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2EZUDFV2AZEKTYO2DVRIFNKKDL","json":"https://pith.science/pith/2EZUDFV2AZEKTYO2DVRIFNKKDL.json","graph_json":"https://pith.science/api/pith-number/2EZUDFV2AZEKTYO2DVRIFNKKDL/graph.json","events_json":"https://pith.science/api/pith-number/2EZUDFV2AZEKTYO2DVRIFNKKDL/events.json","paper":"https://pith.science/paper/2EZUDFV2"},"agent_actions":{"view_html":"https://pith.science/pith/2EZUDFV2AZEKTYO2DVRIFNKKDL","download_json":"https://pith.science/pith/2EZUDFV2AZEKTYO2DVRIFNKKDL.json","view_paper":"https://pith.science/paper/2EZUDFV2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.08615&json=true","fetch_graph":"https://pith.science/api/pith-number/2EZUDFV2AZEKTYO2DVRIFNKKDL/graph.json","fetch_events":"https://pith.science/api/pith-number/2EZUDFV2AZEKTYO2DVRIFNKKDL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2EZUDFV2AZEKTYO2DVRIFNKKDL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2EZUDFV2AZEKTYO2DVRIFNKKDL/action/storage_attestation","attest_author":"https://pith.science/pith/2EZUDFV2AZEKTYO2DVRIFNKKDL/action/author_attestation","sign_citation":"https://pith.science/pith/2EZUDFV2AZEKTYO2DVRIFNKKDL/action/citation_signature","submit_replication":"https://pith.science/pith/2EZUDFV2AZEKTYO2DVRIFNKKDL/action/replication_record"}},"created_at":"2026-07-05T09:49:14.134271+00:00","updated_at":"2026-07-05T09:49:14.134271+00:00"}