{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Z3YHUWECHAILUPMF5KFFWWIHZI","short_pith_number":"pith:Z3YHUWEC","schema_version":"1.0","canonical_sha256":"cef07a58823810ba3d85ea8a5b5907ca2dc1557b77f74b8aaca2111ceb9d5b1a","source":{"kind":"arxiv","id":"2405.20770","version":4},"attestation_state":"computed","paper":{"title":"Large Language Model Sentinel: LLM Agent for Adversarial Purification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR"],"primary_cat":"cs.CL","authors_text":"Guang Lin, Qibin Zhao, Toshihisa Tanaka","submitted_at":"2024-05-24T07:23:56Z","abstract_excerpt":"Over the past two years, the use of large language models (LLMs) has advanced rapidly. While these LLMs offer considerable convenience, they also raise security concerns, as LLMs are vulnerable to adversarial attacks by some well-designed textual perturbations. In this paper, we introduce a novel defense technique named Large LAnguage MOdel Sentinel (LLAMOS), which is designed to enhance the adversarial robustness of LLMs by purifying the adversarial textual examples before feeding them into the target LLM. Our method comprises two main components: a) Agent instruction, which can simulate a ne"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.20770","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-05-24T07:23:56Z","cross_cats_sorted":["cs.AI","cs.CR"],"title_canon_sha256":"74064baa5b25eda053d343eb07b5ba39d263ce4f28e7da82a150e75677a99427","abstract_canon_sha256":"e6b7961b81f30d9736f6fdfe68afef0fe6d869a59a25c8d0a603b072f4d1bb14"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:36.960134Z","signature_b64":"vqy+fDA/VfH/LepAQPmCpy1CN0xRxdtpyzipHvuNEEILSZBKRzJ1p+OG4BoxdNkuxHU5EkNr1BjiQBQrFIktCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cef07a58823810ba3d85ea8a5b5907ca2dc1557b77f74b8aaca2111ceb9d5b1a","last_reissued_at":"2026-07-05T10:52:36.959659Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:36.959659Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Model Sentinel: LLM Agent for Adversarial Purification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR"],"primary_cat":"cs.CL","authors_text":"Guang Lin, Qibin Zhao, Toshihisa Tanaka","submitted_at":"2024-05-24T07:23:56Z","abstract_excerpt":"Over the past two years, the use of large language models (LLMs) has advanced rapidly. While these LLMs offer considerable convenience, they also raise security concerns, as LLMs are vulnerable to adversarial attacks by some well-designed textual perturbations. In this paper, we introduce a novel defense technique named Large LAnguage MOdel Sentinel (LLAMOS), which is designed to enhance the adversarial robustness of LLMs by purifying the adversarial textual examples before feeding them into the target LLM. Our method comprises two main components: a) Agent instruction, which can simulate a ne"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.20770","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.20770/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.20770","created_at":"2026-07-05T10:52:36.959716+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.20770v4","created_at":"2026-07-05T10:52:36.959716+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.20770","created_at":"2026-07-05T10:52:36.959716+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z3YHUWECHAIL","created_at":"2026-07-05T10:52:36.959716+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z3YHUWECHAILUPMF","created_at":"2026-07-05T10:52:36.959716+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z3YHUWEC","created_at":"2026-07-05T10:52:36.959716+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19826","citing_title":"Heterogeneous LLM Debate Under Adversarial Peers: Honest Gains, Replacement Costs, and Resilience","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21460","citing_title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","ref_index":182,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z3YHUWECHAILUPMF5KFFWWIHZI","json":"https://pith.science/pith/Z3YHUWECHAILUPMF5KFFWWIHZI.json","graph_json":"https://pith.science/api/pith-number/Z3YHUWECHAILUPMF5KFFWWIHZI/graph.json","events_json":"https://pith.science/api/pith-number/Z3YHUWECHAILUPMF5KFFWWIHZI/events.json","paper":"https://pith.science/paper/Z3YHUWEC"},"agent_actions":{"view_html":"https://pith.science/pith/Z3YHUWECHAILUPMF5KFFWWIHZI","download_json":"https://pith.science/pith/Z3YHUWECHAILUPMF5KFFWWIHZI.json","view_paper":"https://pith.science/paper/Z3YHUWEC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.20770&json=true","fetch_graph":"https://pith.science/api/pith-number/Z3YHUWECHAILUPMF5KFFWWIHZI/graph.json","fetch_events":"https://pith.science/api/pith-number/Z3YHUWECHAILUPMF5KFFWWIHZI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z3YHUWECHAILUPMF5KFFWWIHZI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z3YHUWECHAILUPMF5KFFWWIHZI/action/storage_attestation","attest_author":"https://pith.science/pith/Z3YHUWECHAILUPMF5KFFWWIHZI/action/author_attestation","sign_citation":"https://pith.science/pith/Z3YHUWECHAILUPMF5KFFWWIHZI/action/citation_signature","submit_replication":"https://pith.science/pith/Z3YHUWECHAILUPMF5KFFWWIHZI/action/replication_record"}},"created_at":"2026-07-05T10:52:36.959716+00:00","updated_at":"2026-07-05T10:52:36.959716+00:00"}