{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YGMYR4RAHRPCZFNBRWWTJDPZHV","short_pith_number":"pith:YGMYR4RA","schema_version":"1.0","canonical_sha256":"c19988f2203c5e2c95a18dad348df93d6d5c4b02a180b9dbcbba4dfecb5875fe","source":{"kind":"arxiv","id":"2406.00799","version":6},"attestation_state":"computed","paper":{"title":"Get my drift? Catching LLM Task Drift with Activation Deltas","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CY"],"primary_cat":"cs.CR","authors_text":"Ahmed Salem, Aideen Fay, Andrew Paverd, Giovanni Cherubin, Mario Fritz, Sahar Abdelnabi","submitted_at":"2024-06-02T16:53:21Z","abstract_excerpt":"LLMs are commonly used in retrieval-augmented applications to execute user instructions based on data from external sources. For example, modern search engines use LLMs to answer queries based on relevant search results; email plugins summarize emails by processing their content through an LLM. However, the potentially untrusted provenance of these data sources can lead to prompt injection attacks, where the LLM is manipulated by natural language instructions embedded in the external data, causing it to deviate from the user's original instruction(s). We define this deviation as task drift. Ta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.00799","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-06-02T16:53:21Z","cross_cats_sorted":["cs.CL","cs.CY"],"title_canon_sha256":"77c52ef8662b47631a5908e596563af99b51f81e56b1e896abc6c7e6b62e4868","abstract_canon_sha256":"c3794f471e498d1c54c56236121f9495edb7ad5a772801385ed35363bbb0abc3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:25:06.913004Z","signature_b64":"MoY+qXy1FgWuVjRj1ZWOuXMoB3QGLE080d6CvQB6VQ9gEbQ12McuxCxZXLmG+hswyvnpfU42Vk3iGPhv3iweBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c19988f2203c5e2c95a18dad348df93d6d5c4b02a180b9dbcbba4dfecb5875fe","last_reissued_at":"2026-07-05T10:25:06.912479Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:25:06.912479Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Get my drift? Catching LLM Task Drift with Activation Deltas","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CY"],"primary_cat":"cs.CR","authors_text":"Ahmed Salem, Aideen Fay, Andrew Paverd, Giovanni Cherubin, Mario Fritz, Sahar Abdelnabi","submitted_at":"2024-06-02T16:53:21Z","abstract_excerpt":"LLMs are commonly used in retrieval-augmented applications to execute user instructions based on data from external sources. For example, modern search engines use LLMs to answer queries based on relevant search results; email plugins summarize emails by processing their content through an LLM. However, the potentially untrusted provenance of these data sources can lead to prompt injection attacks, where the LLM is manipulated by natural language instructions embedded in the external data, causing it to deviate from the user's original instruction(s). We define this deviation as task drift. Ta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.00799","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.00799/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.00799","created_at":"2026-07-05T10:25:06.912534+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.00799v6","created_at":"2026-07-05T10:25:06.912534+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.00799","created_at":"2026-07-05T10:25:06.912534+00:00"},{"alias_kind":"pith_short_12","alias_value":"YGMYR4RAHRPC","created_at":"2026-07-05T10:25:06.912534+00:00"},{"alias_kind":"pith_short_16","alias_value":"YGMYR4RAHRPCZFNB","created_at":"2026-07-05T10:25:06.912534+00:00"},{"alias_kind":"pith_short_8","alias_value":"YGMYR4RA","created_at":"2026-07-05T10:25:06.912534+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06807","citing_title":"When Agents Go Rogue: Activation-Based Detection of Malicious Behaviors in Multi-Agent Systems","ref_index":17,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21843","citing_title":"Measuring What Persists: Conditioning Mechanisms and a Geometric Framework for AI Agent Identity","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00012","citing_title":"PRA-RAG: Provably Robust Aggregation in Retrieval-Augmented Generation against Retrieval Corruption","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29659","citing_title":"Opir: Efficient Multi-Task Safety Classification for Toxicity, Jailbreaks, Hate Speech, and Harmful Content","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2506.17299","citing_title":"Toward Principled LLM Safety Testing: Solving the Jailbreak Oracle Problem","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2603.12230","citing_title":"Security Considerations for Artificial Intelligence Agents","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2404.01318","citing_title":"JailbreakBench: An Open Robustness Benchmark for Jailbreaking Large Language Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12922","citing_title":"When Attention Closes: How LLMs Lose the Thread in Multi-Turn Interaction","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YGMYR4RAHRPCZFNBRWWTJDPZHV","json":"https://pith.science/pith/YGMYR4RAHRPCZFNBRWWTJDPZHV.json","graph_json":"https://pith.science/api/pith-number/YGMYR4RAHRPCZFNBRWWTJDPZHV/graph.json","events_json":"https://pith.science/api/pith-number/YGMYR4RAHRPCZFNBRWWTJDPZHV/events.json","paper":"https://pith.science/paper/YGMYR4RA"},"agent_actions":{"view_html":"https://pith.science/pith/YGMYR4RAHRPCZFNBRWWTJDPZHV","download_json":"https://pith.science/pith/YGMYR4RAHRPCZFNBRWWTJDPZHV.json","view_paper":"https://pith.science/paper/YGMYR4RA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.00799&json=true","fetch_graph":"https://pith.science/api/pith-number/YGMYR4RAHRPCZFNBRWWTJDPZHV/graph.json","fetch_events":"https://pith.science/api/pith-number/YGMYR4RAHRPCZFNBRWWTJDPZHV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YGMYR4RAHRPCZFNBRWWTJDPZHV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YGMYR4RAHRPCZFNBRWWTJDPZHV/action/storage_attestation","attest_author":"https://pith.science/pith/YGMYR4RAHRPCZFNBRWWTJDPZHV/action/author_attestation","sign_citation":"https://pith.science/pith/YGMYR4RAHRPCZFNBRWWTJDPZHV/action/citation_signature","submit_replication":"https://pith.science/pith/YGMYR4RAHRPCZFNBRWWTJDPZHV/action/replication_record"}},"created_at":"2026-07-05T10:25:06.912534+00:00","updated_at":"2026-07-05T10:25:06.912534+00:00"}