{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XN4NTEX2J3L2QDK4Y65CDCGFXQ","short_pith_number":"pith:XN4NTEX2","schema_version":"1.0","canonical_sha256":"bb78d992fa4ed7a80d5cc7ba2188c5bc27cccbf533ca1fd0f668b9efd796be74","source":{"kind":"arxiv","id":"2411.00348","version":2},"attestation_state":"computed","paper":{"title":"Attention Tracker: Detecting Prompt Injection Attacks in LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CR","authors_text":"Ambrish Rawat, Ching-Yun Ko, I-Hsin Chung, Kuo-Han Hung, Pin-Yu Chen, Winston H. Hsu","submitted_at":"2024-11-01T04:05:59Z","abstract_excerpt":"Large Language Models (LLMs) have revolutionized various domains but remain vulnerable to prompt injection attacks, where malicious inputs manipulate the model into ignoring original instructions and executing designated action. In this paper, we investigate the underlying mechanisms of these attacks by analyzing the attention patterns within LLMs. We introduce the concept of the distraction effect, where specific attention heads, termed important heads, shift focus from the original instruction to the injected instruction. Building on this discovery, we propose Attention Tracker, a training-f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.00348","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-11-01T04:05:59Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"b9929bf18b4e02676f72ed93202e4ab3aa24f7895d00cbe4670ac916580acc54","abstract_canon_sha256":"8ba38c5d16518a9891141eb96dc78ce396f78f73e9f411f9ad81f7afa2aed7b4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:38.122561Z","signature_b64":"nzN8Gg+dxSeruRbt3b5ClFoPwjNYzAZw6Dm70Jms5oX1yzy/y+R/WFKWwC5m8hxg7scUBbXNW0t1Dge1qXAYCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb78d992fa4ed7a80d5cc7ba2188c5bc27cccbf533ca1fd0f668b9efd796be74","last_reissued_at":"2026-07-05T10:52:38.122065Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:38.122065Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Attention Tracker: Detecting Prompt Injection Attacks in LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CR","authors_text":"Ambrish Rawat, Ching-Yun Ko, I-Hsin Chung, Kuo-Han Hung, Pin-Yu Chen, Winston H. Hsu","submitted_at":"2024-11-01T04:05:59Z","abstract_excerpt":"Large Language Models (LLMs) have revolutionized various domains but remain vulnerable to prompt injection attacks, where malicious inputs manipulate the model into ignoring original instructions and executing designated action. In this paper, we investigate the underlying mechanisms of these attacks by analyzing the attention patterns within LLMs. We introduce the concept of the distraction effect, where specific attention heads, termed important heads, shift focus from the original instruction to the injected instruction. Building on this discovery, we propose Attention Tracker, a training-f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.00348","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.00348/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.00348","created_at":"2026-07-05T10:52:38.122118+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.00348v2","created_at":"2026-07-05T10:52:38.122118+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.00348","created_at":"2026-07-05T10:52:38.122118+00:00"},{"alias_kind":"pith_short_12","alias_value":"XN4NTEX2J3L2","created_at":"2026-07-05T10:52:38.122118+00:00"},{"alias_kind":"pith_short_16","alias_value":"XN4NTEX2J3L2QDK4","created_at":"2026-07-05T10:52:38.122118+00:00"},{"alias_kind":"pith_short_8","alias_value":"XN4NTEX2","created_at":"2026-07-05T10:52:38.122118+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.02546","citing_title":"To trust or not to trust: Attention-based Trust Management for LLM Multi-Agent Systems","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22040","citing_title":"\"Your AI, My Shell\": Demystifying Prompt Injection Attacks on Agentic AI Coding Editors","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2510.23883","citing_title":"Agentic AI Security: Threats, Defenses, Evaluation, and Open Challenges","ref_index":257,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22888","citing_title":"RouteGuard: Internal-Signal Detection of Skill Poisoning in LLM Agents","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XN4NTEX2J3L2QDK4Y65CDCGFXQ","json":"https://pith.science/pith/XN4NTEX2J3L2QDK4Y65CDCGFXQ.json","graph_json":"https://pith.science/api/pith-number/XN4NTEX2J3L2QDK4Y65CDCGFXQ/graph.json","events_json":"https://pith.science/api/pith-number/XN4NTEX2J3L2QDK4Y65CDCGFXQ/events.json","paper":"https://pith.science/paper/XN4NTEX2"},"agent_actions":{"view_html":"https://pith.science/pith/XN4NTEX2J3L2QDK4Y65CDCGFXQ","download_json":"https://pith.science/pith/XN4NTEX2J3L2QDK4Y65CDCGFXQ.json","view_paper":"https://pith.science/paper/XN4NTEX2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.00348&json=true","fetch_graph":"https://pith.science/api/pith-number/XN4NTEX2J3L2QDK4Y65CDCGFXQ/graph.json","fetch_events":"https://pith.science/api/pith-number/XN4NTEX2J3L2QDK4Y65CDCGFXQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XN4NTEX2J3L2QDK4Y65CDCGFXQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XN4NTEX2J3L2QDK4Y65CDCGFXQ/action/storage_attestation","attest_author":"https://pith.science/pith/XN4NTEX2J3L2QDK4Y65CDCGFXQ/action/author_attestation","sign_citation":"https://pith.science/pith/XN4NTEX2J3L2QDK4Y65CDCGFXQ/action/citation_signature","submit_replication":"https://pith.science/pith/XN4NTEX2J3L2QDK4Y65CDCGFXQ/action/replication_record"}},"created_at":"2026-07-05T10:52:38.122118+00:00","updated_at":"2026-07-05T10:52:38.122118+00:00"}