{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BH2F3ADID3CPCDHFI3DQBA3DQQ","short_pith_number":"pith:BH2F3ADI","schema_version":"1.0","canonical_sha256":"09f45d80681ec4f10ce546c7008363841a7f5caf9eaa5b544fe4216fcecadbef","source":{"kind":"arxiv","id":"2310.10844","version":1},"attestation_state":"computed","paper":{"title":"Survey of Vulnerabilities in Large Language Models Revealed by Adversarial Attacks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Erfan Shayegani, Md Abdullah Al Mamun, Nael Abu-Ghazaleh, Pedram Zaree, Yue Dong, Yu Fu","submitted_at":"2023-10-16T21:37:24Z","abstract_excerpt":"Large Language Models (LLMs) are swiftly advancing in architecture and capability, and as they integrate more deeply into complex systems, the urgency to scrutinize their security properties grows. This paper surveys research in the emerging interdisciplinary field of adversarial attacks on LLMs, a subfield of trustworthy ML, combining the perspectives of Natural Language Processing and Security. Prior work has shown that even safety-aligned LLMs (via instruction tuning and reinforcement learning through human feedback) can be susceptible to adversarial attacks, which exploit weaknesses and mi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.10844","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-16T21:37:24Z","cross_cats_sorted":["cs.CR","cs.LG"],"title_canon_sha256":"ea825d46e71351338acfdfe3848b5a8aa53b7aba3fa59504ed81b4e7d7591eb3","abstract_canon_sha256":"f944763b81bb43fd0cc6c9148e0dae4e33bbbc64025f468329348f3408e22588"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:01:28.600851Z","signature_b64":"g7917keeQNe1da+rWHgSYCgL1AWt337IYuSJiWW5V1F95kCn0p1fNpHPAbTNafyMzC5g50Xocw8oi9s5dCr6Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"09f45d80681ec4f10ce546c7008363841a7f5caf9eaa5b544fe4216fcecadbef","last_reissued_at":"2026-07-05T07:01:28.600371Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:01:28.600371Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Survey of Vulnerabilities in Large Language Models Revealed by Adversarial Attacks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Erfan Shayegani, Md Abdullah Al Mamun, Nael Abu-Ghazaleh, Pedram Zaree, Yue Dong, Yu Fu","submitted_at":"2023-10-16T21:37:24Z","abstract_excerpt":"Large Language Models (LLMs) are swiftly advancing in architecture and capability, and as they integrate more deeply into complex systems, the urgency to scrutinize their security properties grows. This paper surveys research in the emerging interdisciplinary field of adversarial attacks on LLMs, a subfield of trustworthy ML, combining the perspectives of Natural Language Processing and Security. Prior work has shown that even safety-aligned LLMs (via instruction tuning and reinforcement learning through human feedback) can be susceptible to adversarial attacks, which exploit weaknesses and mi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.10844","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.10844/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.10844","created_at":"2026-07-05T07:01:28.600431+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.10844v1","created_at":"2026-07-05T07:01:28.600431+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.10844","created_at":"2026-07-05T07:01:28.600431+00:00"},{"alias_kind":"pith_short_12","alias_value":"BH2F3ADID3CP","created_at":"2026-07-05T07:01:28.600431+00:00"},{"alias_kind":"pith_short_16","alias_value":"BH2F3ADID3CPCDHF","created_at":"2026-07-05T07:01:28.600431+00:00"},{"alias_kind":"pith_short_8","alias_value":"BH2F3ADI","created_at":"2026-07-05T07:01:28.600431+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13439","citing_title":"S-GBT: Smooth Growth Bound Tensor for Certified Robustness Against Word Substitution Attacks in NLP","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07237","citing_title":"When Large Language Models Fail in Healthcare: Evaluating Sensitivity to Prompt Variations","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03024","citing_title":"SkillGuard: A Permission-Centric Framework for Agent Skill Security","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29237","citing_title":"Evolving Skill-Structured Attack Memory Enhances LLM Jailbreaking","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23157","citing_title":"Same Model, Different Weakness: How Language and Modality Reshape the Jailbreak Attack Surface in Frontier MLLMs","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18169","citing_title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","ref_index":132,"is_internal_anchor":false},{"citing_arxiv_id":"2503.06223","citing_title":"RedDiffuser: Auditing Multimodal Safety Failures in Vision-Language Models via Reinforced Diffusion","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2510.16558","citing_title":"A First Look at the Security Issues in the Model Context Protocol Ecosystem","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2510.22628","citing_title":"Sentra-Guard: A Real-Time Multilingual Defense Against Adversarial LLM Prompts","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2512.20677","citing_title":"Learning-Based Automated Adversarial Red-Teaming for Robustness Evaluation of Large Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2404.13501","citing_title":"A Survey on the Memory Mechanism of Large Language Model based Agents","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2501.06322","citing_title":"Multi-Agent Collaboration Mechanisms: A Survey of LLMs","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08898","citing_title":"LLM-Agnostic Semantic Representation Attack","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2502.18864","citing_title":"Towards an AI co-scientist","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07046","citing_title":"An Interpretable and Scalable Framework for Evaluating Large Language Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13803","citing_title":"Gaslight, Gatekeep, V1-V3: Early Visual Cortex Alignment Shields Vision-Language Models from Sycophantic Manipulation","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BH2F3ADID3CPCDHFI3DQBA3DQQ","json":"https://pith.science/pith/BH2F3ADID3CPCDHFI3DQBA3DQQ.json","graph_json":"https://pith.science/api/pith-number/BH2F3ADID3CPCDHFI3DQBA3DQQ/graph.json","events_json":"https://pith.science/api/pith-number/BH2F3ADID3CPCDHFI3DQBA3DQQ/events.json","paper":"https://pith.science/paper/BH2F3ADI"},"agent_actions":{"view_html":"https://pith.science/pith/BH2F3ADID3CPCDHFI3DQBA3DQQ","download_json":"https://pith.science/pith/BH2F3ADID3CPCDHFI3DQBA3DQQ.json","view_paper":"https://pith.science/paper/BH2F3ADI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.10844&json=true","fetch_graph":"https://pith.science/api/pith-number/BH2F3ADID3CPCDHFI3DQBA3DQQ/graph.json","fetch_events":"https://pith.science/api/pith-number/BH2F3ADID3CPCDHFI3DQBA3DQQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BH2F3ADID3CPCDHFI3DQBA3DQQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BH2F3ADID3CPCDHFI3DQBA3DQQ/action/storage_attestation","attest_author":"https://pith.science/pith/BH2F3ADID3CPCDHFI3DQBA3DQQ/action/author_attestation","sign_citation":"https://pith.science/pith/BH2F3ADID3CPCDHFI3DQBA3DQQ/action/citation_signature","submit_replication":"https://pith.science/pith/BH2F3ADID3CPCDHFI3DQBA3DQQ/action/replication_record"}},"created_at":"2026-07-05T07:01:28.600431+00:00","updated_at":"2026-07-05T07:01:28.600431+00:00"}