{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6FMPDCBXIFB3SNL4P63WUWJ5HY","short_pith_number":"pith:6FMPDCBX","schema_version":"1.0","canonical_sha256":"f158f188374143b9357c7fb76a593d3e28875063ac105a30263b778314676cfc","source":{"kind":"arxiv","id":"2402.09154","version":2},"attestation_state":"computed","paper":{"title":"Attacking Large Language Models with Projected Gradient Descent","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Johannes Gasteiger, M. H. I. Abdalla, Simon Geisler, Stephan G\\\"unnemann, Tom Wollschl\\\"ager","submitted_at":"2024-02-14T13:13:26Z","abstract_excerpt":"Current LLM alignment methods are readily broken through specifically crafted adversarial prompts. While crafting adversarial prompts using discrete optimization is highly effective, such attacks typically use more than 100,000 LLM calls. This high computational cost makes them unsuitable for, e.g., quantitative analyses and adversarial training. To remedy this, we revisit Projected Gradient Descent (PGD) on the continuously relaxed input prompt. Although previous attempts with ordinary gradient-based attacks largely failed, we show that carefully controlling the error introduced by the contin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.09154","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-14T13:13:26Z","cross_cats_sorted":[],"title_canon_sha256":"7fe82e64af51279a4c2fd8c93b64da2f08b17729f9c64bbe6c4ef189e402c6e4","abstract_canon_sha256":"04647cc40c6ec86f14d4147814d65b55465ff8907f31b5965d9b3799842ffaf2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:44.171312Z","signature_b64":"OisLSYTF2kWaiaN93yJ5mV56AeishRLSbpXySlnqNydffEXTL+g6hMwdWn6N0PWKtn4nUEjzQsStuPvkVIFrDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f158f188374143b9357c7fb76a593d3e28875063ac105a30263b778314676cfc","last_reissued_at":"2026-07-05T10:22:44.170683Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:44.170683Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Attacking Large Language Models with Projected Gradient Descent","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Johannes Gasteiger, M. H. I. Abdalla, Simon Geisler, Stephan G\\\"unnemann, Tom Wollschl\\\"ager","submitted_at":"2024-02-14T13:13:26Z","abstract_excerpt":"Current LLM alignment methods are readily broken through specifically crafted adversarial prompts. While crafting adversarial prompts using discrete optimization is highly effective, such attacks typically use more than 100,000 LLM calls. This high computational cost makes them unsuitable for, e.g., quantitative analyses and adversarial training. To remedy this, we revisit Projected Gradient Descent (PGD) on the continuously relaxed input prompt. Although previous attempts with ordinary gradient-based attacks largely failed, we show that carefully controlling the error introduced by the contin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.09154","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.09154/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.09154","created_at":"2026-07-05T10:22:44.170770+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.09154v2","created_at":"2026-07-05T10:22:44.170770+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.09154","created_at":"2026-07-05T10:22:44.170770+00:00"},{"alias_kind":"pith_short_12","alias_value":"6FMPDCBXIFB3","created_at":"2026-07-05T10:22:44.170770+00:00"},{"alias_kind":"pith_short_16","alias_value":"6FMPDCBXIFB3SNL4","created_at":"2026-07-05T10:22:44.170770+00:00"},{"alias_kind":"pith_short_8","alias_value":"6FMPDCBX","created_at":"2026-07-05T10:22:44.170770+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00481","citing_title":"Beyond the Prompt: Jailbreaking Function-Calling LLMs via Simulated Moderation Traces","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03647","citing_title":"Black-box, Adaptive, Efficient, Transferable, Harmful, Applicable... Attacks Are All You Need to Break LLMs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2503.02574","citing_title":"LLM-Safety Evaluations Lack Robustness","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09067","citing_title":"Enhancing the Safety of Medical Vision-Language Models by Synthetic Demonstrations","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02356","citing_title":"ASTRA: An Automated Framework for Strategy Discovery, Retrieval, and Evolution for Jailbreaking LLMs","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2404.01318","citing_title":"JailbreakBench: An Open Robustness Benchmark for Jailbreaking Large Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12813","citing_title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","ref_index":158,"is_internal_anchor":false},{"citing_arxiv_id":"2310.03684","citing_title":"SmoothLLM: Defending Large Language Models Against Jailbreaking Attacks","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08898","citing_title":"LLM-Agnostic Semantic Representation Attack","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05116","citing_title":"On the Hardness of Junking LLMs","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09222","citing_title":"GRM: Utility-Aware Jailbreak Attacks on Audio LLMs via Gradient-Ratio Masking","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6FMPDCBXIFB3SNL4P63WUWJ5HY","json":"https://pith.science/pith/6FMPDCBXIFB3SNL4P63WUWJ5HY.json","graph_json":"https://pith.science/api/pith-number/6FMPDCBXIFB3SNL4P63WUWJ5HY/graph.json","events_json":"https://pith.science/api/pith-number/6FMPDCBXIFB3SNL4P63WUWJ5HY/events.json","paper":"https://pith.science/paper/6FMPDCBX"},"agent_actions":{"view_html":"https://pith.science/pith/6FMPDCBXIFB3SNL4P63WUWJ5HY","download_json":"https://pith.science/pith/6FMPDCBXIFB3SNL4P63WUWJ5HY.json","view_paper":"https://pith.science/paper/6FMPDCBX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.09154&json=true","fetch_graph":"https://pith.science/api/pith-number/6FMPDCBXIFB3SNL4P63WUWJ5HY/graph.json","fetch_events":"https://pith.science/api/pith-number/6FMPDCBXIFB3SNL4P63WUWJ5HY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6FMPDCBXIFB3SNL4P63WUWJ5HY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6FMPDCBXIFB3SNL4P63WUWJ5HY/action/storage_attestation","attest_author":"https://pith.science/pith/6FMPDCBXIFB3SNL4P63WUWJ5HY/action/author_attestation","sign_citation":"https://pith.science/pith/6FMPDCBXIFB3SNL4P63WUWJ5HY/action/citation_signature","submit_replication":"https://pith.science/pith/6FMPDCBXIFB3SNL4P63WUWJ5HY/action/replication_record"}},"created_at":"2026-07-05T10:22:44.170770+00:00","updated_at":"2026-07-05T10:22:44.170770+00:00"}