{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:R3MVXNLC4E63YQRO2KEI7KOPF4","short_pith_number":"pith:R3MVXNLC","schema_version":"1.0","canonical_sha256":"8ed95bb562e13dbc422ed2888fa9cf2f2934272bc2f24950fd32e8a228ede85d","source":{"kind":"arxiv","id":"2604.01194","version":2},"attestation_state":"computed","paper":{"title":"AgentWatcher: A Rule-based Prompt Injection Monitor","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Jinyuan Jia, Runpeng Geng, Wei Zou, Yanting Wang","submitted_at":"2026-04-01T17:40:03Z","abstract_excerpt":"Large language models (LLMs) and their applications, such as agents, are highly vulnerable to prompt injection attacks. State-of-the-art prompt injection detection methods have the following limitations: (1) their effectiveness degrades significantly as context length increases, and (2) they lack explicit rules that define what constitutes prompt injection, causing detection decisions to be implicit, opaque, and difficult to reason about. In this work, we propose AgentWatcher to address the above two limitations. To address the first limitation, AgentWatcher attributes the LLM's output (e.g., "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2604.01194","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2026-04-01T17:40:03Z","cross_cats_sorted":[],"title_canon_sha256":"4319748e955d1031d3a94558c6d590f6e88e693584f39e2638d0b5c332bc18e1","abstract_canon_sha256":"1738a1314d9110fa51299dcdc3e9001dbaeaada0a5c10cdab45481fc116dd0fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-28T01:22:32.691122Z","signature_b64":"4s7AS96iAIaUMUDMJqq541vgrbjB2SlIKFjS/MZ+kIoKAZIAWe01CY1U+8nDCr9ph2684av5r8BxW5aBB4PqAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8ed95bb562e13dbc422ed2888fa9cf2f2934272bc2f24950fd32e8a228ede85d","last_reissued_at":"2026-07-28T01:22:32.690164Z","signature_status":"signed_v1","first_computed_at":"2026-07-28T01:22:32.690164Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AgentWatcher: A Rule-based Prompt Injection Monitor","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Jinyuan Jia, Runpeng Geng, Wei Zou, Yanting Wang","submitted_at":"2026-04-01T17:40:03Z","abstract_excerpt":"Large language models (LLMs) and their applications, such as agents, are highly vulnerable to prompt injection attacks. State-of-the-art prompt injection detection methods have the following limitations: (1) their effectiveness degrades significantly as context length increases, and (2) they lack explicit rules that define what constitutes prompt injection, causing detection decisions to be implicit, opaque, and difficult to reason about. In this work, we propose AgentWatcher to address the above two limitations. To address the first limitation, AgentWatcher attributes the LLM's output (e.g., "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2604.01194","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.01194/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2604.01194","created_at":"2026-07-28T01:22:32.690623+00:00"},{"alias_kind":"arxiv_version","alias_value":"2604.01194v2","created_at":"2026-07-28T01:22:32.690623+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.01194","created_at":"2026-07-28T01:22:32.690623+00:00"},{"alias_kind":"pith_short_12","alias_value":"R3MVXNLC4E63","created_at":"2026-07-28T01:22:32.690623+00:00"},{"alias_kind":"pith_short_16","alias_value":"R3MVXNLC4E63YQRO","created_at":"2026-07-28T01:22:32.690623+00:00"},{"alias_kind":"pith_short_8","alias_value":"R3MVXNLC","created_at":"2026-07-28T01:22:32.690623+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2605.17453","citing_title":"Trust No Tool: Evaluating and Defending LLM Agents under Untrusted Tool Feedback","ref_index":36,"is_internal_anchor":true},{"citing_arxiv_id":"2603.28013","citing_title":"Kill-Chain Canaries: Stage-Level Tracking of Prompt Injection Across Attack Surfaces and Model Safety Tiers","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R3MVXNLC4E63YQRO2KEI7KOPF4","json":"https://pith.science/pith/R3MVXNLC4E63YQRO2KEI7KOPF4.json","graph_json":"https://pith.science/api/pith-number/R3MVXNLC4E63YQRO2KEI7KOPF4/graph.json","events_json":"https://pith.science/api/pith-number/R3MVXNLC4E63YQRO2KEI7KOPF4/events.json","paper":"https://pith.science/paper/R3MVXNLC"},"agent_actions":{"view_html":"https://pith.science/pith/R3MVXNLC4E63YQRO2KEI7KOPF4","download_json":"https://pith.science/pith/R3MVXNLC4E63YQRO2KEI7KOPF4.json","view_paper":"https://pith.science/paper/R3MVXNLC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2604.01194&json=true","fetch_graph":"https://pith.science/api/pith-number/R3MVXNLC4E63YQRO2KEI7KOPF4/graph.json","fetch_events":"https://pith.science/api/pith-number/R3MVXNLC4E63YQRO2KEI7KOPF4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R3MVXNLC4E63YQRO2KEI7KOPF4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R3MVXNLC4E63YQRO2KEI7KOPF4/action/storage_attestation","attest_author":"https://pith.science/pith/R3MVXNLC4E63YQRO2KEI7KOPF4/action/author_attestation","sign_citation":"https://pith.science/pith/R3MVXNLC4E63YQRO2KEI7KOPF4/action/citation_signature","submit_replication":"https://pith.science/pith/R3MVXNLC4E63YQRO2KEI7KOPF4/action/replication_record"}},"created_at":"2026-07-28T01:22:32.690623+00:00","updated_at":"2026-07-28T01:22:32.690623+00:00"}