{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5WEWT3CPQH3ECG55BMYGFJHGSD","short_pith_number":"pith:5WEWT3CP","schema_version":"1.0","canonical_sha256":"ed8969ec4f81f6411bbd0b3062a4e690dedef9f3b796d4cab904bafbe216631e","source":{"kind":"arxiv","id":"2412.13705","version":1},"attestation_state":"computed","paper":{"title":"Mitigating Adversarial Attacks in LLMs through Defensive Suffix Generation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Byeolhee Kim, Gaeun Kee, Heejung Choi, Hyeram Seo, HyoJe Jung, JiYe Han, Minkyoung Kim, Sanghyun Park, Soyoung Ko, Tae Joon Jun, Young-Hak Kim, Yunha Kim","submitted_at":"2024-12-18T10:49:41Z","abstract_excerpt":"Large language models (LLMs) have exhibited outstanding performance in natural language processing tasks. However, these models remain susceptible to adversarial attacks in which slight input perturbations can lead to harmful or misleading outputs. A gradient-based defensive suffix generation algorithm is designed to bolster the robustness of LLMs. By appending carefully optimized defensive suffixes to input prompts, the algorithm mitigates adversarial influences while preserving the models' utility. To enhance adversarial understanding, a novel total loss function ($L_{\\text{total}}$) combini"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.13705","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-18T10:49:41Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"4fe0187fc401f7fd2afd2876429a64a78388a2f8fffc72a973c7ba5056302256","abstract_canon_sha256":"e60e9a4b030223126686ce7be18069d4d87fc6caab0388ab7f37bcb74aac9519"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:51:06.714044Z","signature_b64":"G+TeZEbNpSadwO+e3vQLDoFYqqK6cosj9lfdj8+V2FQ/0icgkQiDWCZRdAT3g6MD4FCkDagAxzfBejPRmRDgBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ed8969ec4f81f6411bbd0b3062a4e690dedef9f3b796d4cab904bafbe216631e","last_reissued_at":"2026-07-05T09:51:06.713555Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:51:06.713555Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mitigating Adversarial Attacks in LLMs through Defensive Suffix Generation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Byeolhee Kim, Gaeun Kee, Heejung Choi, Hyeram Seo, HyoJe Jung, JiYe Han, Minkyoung Kim, Sanghyun Park, Soyoung Ko, Tae Joon Jun, Young-Hak Kim, Yunha Kim","submitted_at":"2024-12-18T10:49:41Z","abstract_excerpt":"Large language models (LLMs) have exhibited outstanding performance in natural language processing tasks. However, these models remain susceptible to adversarial attacks in which slight input perturbations can lead to harmful or misleading outputs. A gradient-based defensive suffix generation algorithm is designed to bolster the robustness of LLMs. By appending carefully optimized defensive suffixes to input prompts, the algorithm mitigates adversarial influences while preserving the models' utility. To enhance adversarial understanding, a novel total loss function ($L_{\\text{total}}$) combini"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.13705","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.13705/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.13705","created_at":"2026-07-05T09:51:06.713615+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.13705v1","created_at":"2026-07-05T09:51:06.713615+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.13705","created_at":"2026-07-05T09:51:06.713615+00:00"},{"alias_kind":"pith_short_12","alias_value":"5WEWT3CPQH3E","created_at":"2026-07-05T09:51:06.713615+00:00"},{"alias_kind":"pith_short_16","alias_value":"5WEWT3CPQH3ECG55","created_at":"2026-07-05T09:51:06.713615+00:00"},{"alias_kind":"pith_short_8","alias_value":"5WEWT3CP","created_at":"2026-07-05T09:51:06.713615+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5WEWT3CPQH3ECG55BMYGFJHGSD","json":"https://pith.science/pith/5WEWT3CPQH3ECG55BMYGFJHGSD.json","graph_json":"https://pith.science/api/pith-number/5WEWT3CPQH3ECG55BMYGFJHGSD/graph.json","events_json":"https://pith.science/api/pith-number/5WEWT3CPQH3ECG55BMYGFJHGSD/events.json","paper":"https://pith.science/paper/5WEWT3CP"},"agent_actions":{"view_html":"https://pith.science/pith/5WEWT3CPQH3ECG55BMYGFJHGSD","download_json":"https://pith.science/pith/5WEWT3CPQH3ECG55BMYGFJHGSD.json","view_paper":"https://pith.science/paper/5WEWT3CP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.13705&json=true","fetch_graph":"https://pith.science/api/pith-number/5WEWT3CPQH3ECG55BMYGFJHGSD/graph.json","fetch_events":"https://pith.science/api/pith-number/5WEWT3CPQH3ECG55BMYGFJHGSD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5WEWT3CPQH3ECG55BMYGFJHGSD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5WEWT3CPQH3ECG55BMYGFJHGSD/action/storage_attestation","attest_author":"https://pith.science/pith/5WEWT3CPQH3ECG55BMYGFJHGSD/action/author_attestation","sign_citation":"https://pith.science/pith/5WEWT3CPQH3ECG55BMYGFJHGSD/action/citation_signature","submit_replication":"https://pith.science/pith/5WEWT3CPQH3ECG55BMYGFJHGSD/action/replication_record"}},"created_at":"2026-07-05T09:51:06.713615+00:00","updated_at":"2026-07-05T09:51:06.713615+00:00"}