{"as_of":"2026-08-08T11:03:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:355829a2d9250cdcae146a51aa38e7ee6a35fa4db88934900f9eead07ff07a96","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":37,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T00:38:15.346908Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":14,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2402.06922","last_updated":"2026-04-21T14:06:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-10T11:07:24Z","title":"Whispers in the Machine: Confidentiality in Agentic Systems","version":5},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-24T03:59:03.972043Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2402.06922"},"observation_digest":"sha256:7d42ce930b5974f2731d3466d294bcc761b5031e1e1637d4a40b82607bb3ae4e","observation_id":"8bcc565e-2fd1-4a78-95d7-86c2094b28be","resolution":{"observed_at":"2026-05-24T04:03:53.924075Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2408.12935","last_updated":"2026-05-13T07:56:42Z","snapshot_observed_at":"2026-08-02T12:48:59.218457Z","submitted_at":"2024-08-23T09:33:48Z","title":"AI Safety Landscape for Large Language Models: Taxonomy, State-of-the-art, and Future Directions","version":4},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-23T21:54:26.670284Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2408.12935"},"observation_digest":"sha256:69d8162ac5be2a9720ad6e7321cb2fd1bb09507c71c29d2f9d134bdf1d741f5c","observation_id":"ba6f02e6-5b5e-4765-a2d5-7ce7ab6c3470","resolution":{"observed_at":"2026-05-23T21:55:50.756346Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2502.05206","last_updated":"2026-04-14T16:10:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-02T05:14:22Z","title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","version":6},"reference_index":293,"source":"pdf_text","source_observed_at":"2026-05-23T04:39:04.591722Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2502.05206"},"observation_digest":"sha256:87347360b842b7343bf14536a84120f320661c86759d4685e4acdd133549d66f","observation_id":"2586924b-d8a2-40a5-8ac9-2d56e349ac96","resolution":{"observed_at":"2026-05-23T04:42:34.252950Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-07T19:45:19.262296Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.14881","last_updated":"2025-02-14T08:42:43Z","snapshot_observed_at":"2026-08-07T22:02:21.919237Z","submitted_at":"2025-02-14T08:42:43Z","title":"A Survey of Safety on Large Vision-Language Models: Attacks, Defenses and Evaluations","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-07T19:45:19.262296Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2502.14881"},"observation_digest":"sha256:79e2557690b8207ec68942bb519aed2a0fb32e3e9b7cb4debc1236cff3a9250e","observation_id":"5fbed452-07e8-4d5e-bcb8-94d749fcab45","resolution":{"observed_at":"2026-08-07T19:45:19.262296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2503.06223","last_updated":"2026-08-06T06:47:13Z","snapshot_observed_at":"2026-08-08T10:14:34.137582Z","submitted_at":"2025-03-08T13:51:40Z","title":"RedDiffuser: Auditing Multimodal Safety Failures in Vision-Language Models via Reinforced Diffusion","version":5},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-23T00:13:08.603115Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2503.06223"},"observation_digest":"sha256:b940fa4dbf355283dac64e8cc4f1acd48c24a5b7e2a8e27070a59a004ed361f8","observation_id":"72c7d2c3-637d-49f7-9744-6d1eec08e422","resolution":{"observed_at":"2026-05-23T00:15:14.829234Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-07T13:05:34.852642Z","title":"(ab)using images and sounds for indirect instruction injection in multi-modal llms.CoRR, abs/2307.10490, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23847","last_updated":"2026-07-06T06:42:57Z","snapshot_observed_at":"2026-08-07T12:58:56.457220Z","submitted_at":"2025-05-28T18:19:03Z","title":"Seven Security Challenges in Cross-domain Multi-agent LLM Systems","version":5},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:34.852642Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2505.23847"},"observation_digest":"sha256:1df3eef5726248ce283642401936d6aaeb1ecc4513ce2d96a2f36eee8d94fbe0","observation_id":"e177162d-e68a-4e9f-a1e1-382d99dbc1e8","resolution":{"observed_at":"2026-08-07T13:05:34.852642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-07T12:07:58.079163Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00548","last_updated":"2025-05-31T13:11:14Z","snapshot_observed_at":"2026-08-08T10:05:29.642578Z","submitted_at":"2025-05-31T13:11:14Z","title":"Con Instruction: Universal Jailbreaking of Multimodal Large Language Models via Non-Textual Modalities","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T12:07:58.079163Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2506.00548"},"observation_digest":"sha256:0f7cdb12dfc048234038135ecdb8ec5ac2189f46e38ba136f78aab381087aa6a","observation_id":"e45d494d-1e18-483f-80c2-3892c392dc02","resolution":{"observed_at":"2026-08-07T12:07:58.079163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-07T10:40:54.417585Z","title":"(2023, February)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.04679","last_updated":"2025-06-05T06:57:28Z","snapshot_observed_at":"2026-08-07T20:31:12.431523Z","submitted_at":"2025-06-05T06:57:28Z","title":"Normative Conflicts and Shallow AI Alignment","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T10:40:54.417585Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2506.04679"},"observation_digest":"sha256:34f7dd0059343106c596ac412f26f5680a717e694f6aba1b0a5d23605fffa910","observation_id":"75da2dfd-2a96-4e97-9eda-f29973741e30","resolution":{"observed_at":"2026-08-07T10:40:54.417585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-06T16:34:05.101373Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.13169","last_updated":"2025-07-17T14:33:36Z","snapshot_observed_at":"2026-08-07T22:34:49.318621Z","submitted_at":"2025-07-17T14:33:36Z","title":"Prompt Injection 2.0: Hybrid AI Threats","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T16:34:05.101373Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2507.13169"},"observation_digest":"sha256:0f4c0f0ce72738e95294083793265b2b6ac9fdf0fb48198c878a702acf331d4e","observation_id":"80107157-42c9-4510-a17b-5ec0990ce669","resolution":{"observed_at":"2026-08-06T16:34:05.101373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-06T12:09:39.603513Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.22037","last_updated":"2025-07-29T17:39:48Z","snapshot_observed_at":"2026-08-06T15:54:16.679483Z","submitted_at":"2025-07-29T17:39:48Z","title":"Secure Tug-of-War (SecTOW): Iterative Defense-Attack Training with Reinforcement Learning for Multimodal Model Security","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T12:09:39.603513Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2507.22037"},"observation_digest":"sha256:cfc0da3a7b5e883a1e80c9c19e18f15bc03375a24c2c8c40f4e98e49e2670d8b","observation_id":"6cd493d6-d804-492f-b72c-8cde5d1c5bb4","resolution":{"observed_at":"2026-08-06T12:09:39.603513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T20:31:43.258454Z","title":"Abusing images and sounds for indirect instruction injection in multi-modal llms","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.10404","last_updated":"2025-08-14T07:12:44Z","snapshot_observed_at":"2026-08-08T02:31:17.622343Z","submitted_at":"2025-08-14T07:12:44Z","title":"Layer-Wise Perturbations via Sparse Autoencoders for Adversarial Text Generation","version":1},"reference_index":111,"source":"pdf_text","source_observed_at":"2026-08-05T20:31:43.258454Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2508.10404"},"observation_digest":"sha256:3f3dbc57f72fff5cc6f638eee3e057269a28245b487d3f5ccb4ea87f1ec454ed","observation_id":"f12926ff-74ab-4146-9cad-112b3fb9769a","resolution":{"observed_at":"2026-08-05T20:31:43.258454Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T19:39:09.526632Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.12175","last_updated":"2025-08-16T22:56:51Z","snapshot_observed_at":"2026-08-06T23:10:25.204698Z","submitted_at":"2025-08-16T22:56:51Z","title":"Invitation Is All You Need! Promptware Attacks Against LLM-Powered Assistants in Production Are Practical and Dangerous","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T19:39:09.526632Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2508.12175"},"observation_digest":"sha256:4dd5473292c05c7df473f281303388b0590ff9e7344ad0bff523a558074fd5f8","observation_id":"b451794c-2a09-460f-9695-11d6ad1a2e58","resolution":{"observed_at":"2026-08-05T19:39:09.526632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2510.23883","last_updated":"2026-04-03T16:27:34Z","snapshot_observed_at":"2026-08-02T13:42:34.526072Z","submitted_at":"2025-10-27T21:48:11Z","title":"Agentic AI Security: Threats, Defenses, Evaluation, and Open Challenges","version":3},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-18T03:42:10.703369Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2510.23883"},"observation_digest":"sha256:0e3caca28c309cb4fdca2c779ee303a14f74ab466e7a8a8c9fb5330b1f62cba3","observation_id":"d2734992-083f-4ae0-b78f-89aaeff0a021","resolution":{"observed_at":"2026-05-18T03:42:22.045980Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-04T07:01:14.163930Z","title":"Bagdasaryan, T.-Y","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.27275","last_updated":"2026-07-30T09:01:41Z","snapshot_observed_at":"2026-08-08T09:17:42.786697Z","submitted_at":"2025-10-31T08:35:42Z","title":"Prevalence of Security and Privacy Risk-Inducing Usage of AI-based Conversational Agents","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T07:01:14.163930Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2510.27275"},"observation_digest":"sha256:7716197aadcf625b70a4efd7eee960f19a6f7b20b33458a4853f06659f25e4f1","observation_id":"f42a9e4d-5ec1-476b-9dd8-94feee168b2e","resolution":{"observed_at":"2026-08-04T07:01:14.163930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2512.12069","last_updated":"2026-04-20T07:36:16Z","snapshot_observed_at":"2026-08-07T10:08:39.637438Z","submitted_at":"2025-12-12T22:31:38Z","title":"Rethinking Jailbreak Detection of Large Vision Language Models with Representational Contrastive Scoring","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-16T22:28:41.134253Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2512.12069"},"observation_digest":"sha256:e3059d5e968c61899530f2f5525de3eeba1413cbf08de1ea504271b504cb39f4","observation_id":"8463fab9-571f-4e8e-b7dd-dabc3bfac939","resolution":{"observed_at":"2026-05-16T22:31:19.422099Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-07-12T20:09:57.922788Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2604.14603","last_updated":"2026-06-29T06:44:20Z","snapshot_observed_at":"2026-07-12T20:09:54.804124Z","submitted_at":"2026-04-16T04:21:32Z","title":"A Synonymous Variational Perspective on the Rate-Distortion-Perception Tradeoff","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-12T20:09:57.922788Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2604.14603"},"observation_digest":"sha256:bdba15920a70330f454ba94ee046c2e7c8ced07b78573d8a2022908dce2caca6","observation_id":"34480220-1d5d-4fee-98a5-820723d20d35","resolution":{"observed_at":"2026-07-12T20:09:57.922788Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2604.14604","last_updated":"2026-04-16T04:22:11Z","snapshot_observed_at":"2026-08-01T20:57:25.785838Z","submitted_at":"2026-04-16T04:22:11Z","title":"Hijacking Large Audio-Language Models via Context-Agnostic and Imperceptible Auditory Prompt Injection","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T11:32:10.126062Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2604.14604"},"observation_digest":"sha256:5aa7e0b7ef7f3292a7e3092352c7ab62fb6b16117020279b8ee79924c2dd52e9","observation_id":"ba77251d-9df2-4d85-8ca2-bdeb9251ed60","resolution":{"observed_at":"2026-05-10T11:35:18.867051Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2604.18860","last_updated":"2026-04-20T21:36:16Z","snapshot_observed_at":"2026-07-06T23:05:39.724237Z","submitted_at":"2026-04-20T21:36:16Z","title":"Temporal UI State Inconsistency in Desktop GUI Agents: Formalizing and Defending Against TOCTOU Attacks on Computer-Use Agents","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T03:51:47.820649Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2604.18860"},"observation_digest":"sha256:3ec9e7e40f6532139051eed8091590abe29573927bb17f76617757d385b2baac","observation_id":"e7e78990-4ca0-4ece-913f-9385457e017b","resolution":{"observed_at":"2026-05-11T12:21:04.276860Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2604.21477","last_updated":"2026-07-14T16:11:06Z","snapshot_observed_at":"2026-08-05T05:24:49.455576Z","submitted_at":"2026-04-23T09:39:15Z","title":"MCP Pitfall Lab: Exposing Developer Pitfalls in MCP Tool Server Security under Multi-Vector Attacks","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-09T21:37:28.671785Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2604.21477"},"observation_digest":"sha256:a5d209243c75157ff33e18eb338aac286231fc534b4fc7c56783ea867d818916","observation_id":"f7c90804-d766-4002-a5b8-6af4d28cd976","resolution":{"observed_at":"2026-05-11T14:31:07.944342Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2604.23374","last_updated":"2026-04-25T16:39:16Z","snapshot_observed_at":"2026-07-06T23:09:38.799345Z","submitted_at":"2026-04-25T16:39:16Z","title":"Ghost in the Agent: Redefining Information Flow Tracking for LLM Agents","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-08T08:08:24.524671Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2604.23374"},"observation_digest":"sha256:49518f6571c92e2eee0a4c01e40fe19fcb9d1f674633b252eabce178009a8ec4","observation_id":"3d59fd30-10a5-4c33-a868-42015cdb1a8f","resolution":{"observed_at":"2026-05-08T22:39:20.664808Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2604.24790","last_updated":"2026-04-25T10:52:29Z","snapshot_observed_at":"2026-07-06T23:10:47.277692Z","submitted_at":"2026-04-25T10:52:29Z","title":"Semantic Denial of Service in LLM-controlled robots","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-08T07:59:42.478294Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2604.24790"},"observation_digest":"sha256:c12b34ed4b6def08629598cfc86d2577b26fe1aa5cbdba09bb6c35512ddc6ed0","observation_id":"86b07d54-192f-42a8-bf58-7f0780903955","resolution":{"observed_at":"2026-05-11T20:46:14.602650Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2604.27267","last_updated":"2026-05-04T05:54:12Z","snapshot_observed_at":"2026-07-06T23:12:47.745700Z","submitted_at":"2026-04-29T23:44:07Z","title":"From Prompt to Physical Actuation: Holistic Threat Modeling of LLM-Enabled Robotic Systems","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-07T09:44:14.993126Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2604.27267"},"observation_digest":"sha256:b53319f963ae53dda95981455a3b09027ac65ba371352e72d1acfee7b0002f9e","observation_id":"2bf34282-1d7d-4985-99ff-d4341cef654e","resolution":{"observed_at":"2026-05-12T09:41:26.904379Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2605.01449","last_updated":"2026-05-02T13:56:50Z","snapshot_observed_at":"2026-08-03T12:34:05.607551Z","submitted_at":"2026-05-02T13:56:50Z","title":"VisInject: Disruption != Injection -- A Dual-Dimension Evaluation of Universal Adversarial Attacks on Vision-Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-09T14:24:48.999632Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2605.01449"},"observation_digest":"sha256:d061e6e2592d01214bbcd2dcb26de99225d58c59dfe96fe285dd4ae1900ca752","observation_id":"8b73f468-d843-4c22-a309-b934f46880ed","resolution":{"observed_at":"2026-05-11T16:56:08.007157Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2605.04261","last_updated":"2026-05-05T19:55:28Z","snapshot_observed_at":"2026-07-06T23:17:04.200299Z","submitted_at":"2026-05-05T19:55:28Z","title":"Laundering AI Authority with Adversarial Examples","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-08T17:19:38.662062Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2605.04261"},"observation_digest":"sha256:003f6d505bb331b2ceb563f945d5a596ef9e412cf170871fe74d65f4d5263d6e","observation_id":"dd5b62e4-8956-45c1-8185-8601b0578196","resolution":{"observed_at":"2026-05-11T17:41:08.017954Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2605.07490","last_updated":"2026-05-08T09:29:50Z","snapshot_observed_at":"2026-07-06T23:19:51.414344Z","submitted_at":"2026-05-08T09:29:50Z","title":"Cross-Modal Backdoors in Multimodal Large Language Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-11T01:51:11.432424Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2605.07490"},"observation_digest":"sha256:6af8f9ccf394a5afe4276e7fb22f1f234861c1a275a206fea2f55e7652735168","observation_id":"b3141f92-532f-4220-9c54-27138fe399e7","resolution":{"observed_at":"2026-05-11T04:20:55.541381Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2605.19192","last_updated":"2026-05-20T21:49:08Z","snapshot_observed_at":"2026-08-03T06:06:31.296336Z","submitted_at":"2026-05-18T23:40:43Z","title":"Hallucination as Exploit: Evidence-Carrying Multimodal Agents","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-20T09:46:42.413501Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2605.19192"},"observation_digest":"sha256:3a148b5e767dbfec3438a9de49673a56e0a5805897ac94b8b81b8a3254bda52a","observation_id":"a11d1780-b5a2-4e01-8618-3e765bf64265","resolution":{"observed_at":"2026-05-20T09:48:11.621232Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2605.19192","last_updated":"2026-05-20T21:49:08Z","snapshot_observed_at":"2026-08-03T06:06:31.296336Z","submitted_at":"2026-05-18T23:40:43Z","title":"Hallucination as Exploit: Evidence-Carrying Multimodal Agents","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-22T08:57:29.491043Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2605.19192"},"observation_digest":"sha256:159e870bd2445af6e1849900cc5a66c9056c8eb3d2fe92ff13f2cb216ba56814","observation_id":"37c91339-e6ee-4f0d-a953-64b7410774ae","resolution":{"observed_at":"2026-05-22T09:01:19.902347Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2605.30454","last_updated":"2026-05-28T18:26:40Z","snapshot_observed_at":"2026-07-31T16:31:43.334849Z","submitted_at":"2026-05-28T18:26:40Z","title":"The Surface You Test Is Not the Surface That Breaks","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-06-29T06:37:19.674012Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2605.30454"},"observation_digest":"sha256:5c2a537bc308d70920b9ee4e776befe9f4eb441b8acf47d15add98141867a5dc","observation_id":"b2cc4d12-f2ca-4f45-ae83-c5cd23ffa2c7","resolution":{"observed_at":"2026-06-29T14:33:31.206657Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2606.02449","last_updated":"2026-06-01T16:20:45Z","snapshot_observed_at":"2026-07-06T23:42:49.173711Z","submitted_at":"2026-06-01T16:20:45Z","title":"HLL: Can Agents Cross Humanity's Last Line of Verification?","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-28T14:57:57.218669Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2606.02449"},"observation_digest":"sha256:54650c9c55fee4e2e219a00122275438889d9c4c77c1a5c4953988d4f8df9300","observation_id":"f1131dda-ded6-43fc-b755-4e2713447359","resolution":{"observed_at":"2026-07-01T22:56:19.607755Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2606.03793","last_updated":"2026-06-02T15:42:10Z","snapshot_observed_at":"2026-08-05T09:22:21.521124Z","submitted_at":"2026-06-02T15:42:10Z","title":"Exploring Adversarial Robustness and Safety Alignment in Multilingual Multi-Modal Large Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-28T10:10:38.261635Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2606.03793"},"observation_digest":"sha256:dec6d2928e5a2a1b821005c733bde2aa78e0a5a9787aaf592713c64f0d68ffae","observation_id":"10cc6aee-624b-410a-a368-a9a0c102c966","resolution":{"observed_at":"2026-07-02T03:16:34.719751Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":"2307.10490","doi":"10.48550/arxiv.2307.10490","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"(ab) using images and sounds for indirect instruction injection in multi-modal llms","venue":"arXiv (Cornell University)","work_id":"cbf07be0-59b9-485c-a352-3971f60f3d56","year":2023},"citing_paper":{"arxiv_id":"2606.05976","last_updated":"2026-07-31T08:26:14Z","snapshot_observed_at":"2026-08-05T23:10:45.588490Z","submitted_at":"2026-06-04T10:17:00Z","title":"The Self-Correction Illusion: Role Relabeling Gates Explicit Error Flagging in Large Language Models","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-06-28T01:25:07.890796Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2606.05976"},"observation_digest":"sha256:943e5c9bd81c7328062ab1ced3f5540b339ba0c78993d4dd4566f5178850eb73","observation_id":"acb5163e-3814-4a3b-85b2-f480b497ffa0","resolution":{"observed_at":"2026-06-28T01:31:29.301820Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-03T02:14:12.243970Z","title":"arXiv:2307.10490","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.05976","last_updated":"2026-07-31T08:26:14Z","snapshot_observed_at":"2026-08-05T23:10:45.588490Z","submitted_at":"2026-06-04T10:17:00Z","title":"The Self-Correction Illusion: Role Relabeling Gates Explicit Error Flagging in Large Language Models","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-03T02:14:12.243970Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2606.05976"},"observation_digest":"sha256:b0f3557eaee67d2319858bd2f8fef86d42132571e61b8642651886c618332ae2","observation_id":"bfa57969-40fc-4e2e-8e8c-8cc1183fea77","resolution":{"observed_at":"2026-08-03T02:14:12.243970Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-07-14T13:02:42.673767Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.10269","last_updated":"2026-07-11T11:56:05Z","snapshot_observed_at":"2026-08-06T13:53:30.569509Z","submitted_at":"2026-07-11T11:56:05Z","title":"Devil in the Lens: Analyzing and Defending Physical Prompt Injection Against Vision-Language Models on Wearable Devices","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-14T13:02:42.673767Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2607.10269"},"observation_digest":"sha256:26018915f6dc1b7e55c853362f76e3f7dc5f09ad7ffdb212b1d8fe65dda71fdc","observation_id":"e9c648c7-be23-4f06-805c-ed9d7d789a84","resolution":{"observed_at":"2026-07-14T13:02:42.673767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-01T22:43:50.536410Z","title":"Abusing images and sounds for indirect instruction injection in multi-modal llms.arXiv preprint arXiv:2307.10490, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.15657","last_updated":"2026-07-17T06:05:17Z","snapshot_observed_at":"2026-08-08T02:14:12.766320Z","submitted_at":"2026-07-17T06:05:17Z","title":"Do Agents Dream of False Memories? Black-box Visual Attacks on Long-term Memory in Multimodal AI Agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-01T22:43:50.536410Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2607.15657"},"observation_digest":"sha256:31f25f63807e35ae95113bb6de0746c0d8d62b6b6174ed079511fc40ac89ec51","observation_id":"3ca2b105-d4f9-4efc-a76a-fcd377754eae","resolution":{"observed_at":"2026-08-01T22:43:50.536410Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-01T06:04:46.731247Z","title":"arXiv preprint arXiv:2307.10490 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22024","last_updated":"2026-07-24T06:43:09Z","snapshot_observed_at":"2026-08-06T15:37:49.497100Z","submitted_at":"2026-07-24T06:43:09Z","title":"Agent Security Needs Redefinition through a Holistic Framework","version":1},"reference_index":295,"source":"arxiv_source","source_observed_at":"2026-08-01T06:04:46.731247Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2607.22024"},"observation_digest":"sha256:0b79b328f9fe5be767b15cda137cc257d9a0f82198a38ae5f0b91b901bcf1361","observation_id":"9b7750ee-0b58-4992-bd03-9af838bcdeb8","resolution":{"observed_at":"2026-08-01T06:04:46.731247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-06T00:20:22.213214Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.01373","last_updated":"2026-08-02T16:49:25Z","snapshot_observed_at":"2026-08-07T07:08:27.911115Z","submitted_at":"2026-08-02T16:49:25Z","title":"The Boy Who Cried Wolf: Adversarial Misclassification of Safe Inputs as Unsafe in Multimodal Guardrails","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T00:20:22.213214Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2608.01373"},"observation_digest":"sha256:9ef93b2e3c44e04ebd5446f00518b4b0703208699607ecd8df4637ba07a86cfe","observation_id":"293a4c02-1cd1-473b-8c3d-18967550e9d9","resolution":{"observed_at":"2026-08-06T00:20:22.213214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10490","snapshot_observed_at":"2026-08-08T00:38:15.346908Z","title":"arXiv preprint arXiv:2307.10490 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.05715","last_updated":"2026-08-06T07:52:55Z","snapshot_observed_at":"2026-08-08T10:15:39.591091Z","submitted_at":"2026-08-06T07:52:55Z","title":"Hijacking Robots with a Piece of Paper: A Systematic Study of Physical Prompt Injection in VLM-Controlled Robots","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-08T00:38:15.346908Z"},"links":{"cited_paper":"/paper/2307.10490","citing_paper":"/paper/2608.05715"},"observation_digest":"sha256:3e73361d57eaef64931761e79599fa022307878dd5b297df3ae112c52bfd3e5f","observation_id":"6e5903d9-8340-4ce3-87d2-e302054779fd","resolution":{"observed_at":"2026-08-08T00:38:15.346908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2307.10490/citation-record","integrity":"/paper/2307.10490/integrity","json":"/paper/2307.10490/citation-record.json","paper":"/paper/2307.10490"},"outbound":[],"paper":{"arxiv_id":"2307.10490","last_updated":"2023-10-03T17:03:10Z","latest_version":4,"primary_category":"cs.CR","snapshot_observed_at":"2026-08-04T02:19:57.679295Z","submitted_at":"2023-07-19T23:03:20Z","title":"Abusing Images and Sounds for Indirect Instruction Injection in Multi-Modal LLMs"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 37 inbound Pith citation observations for arXiv:2307.10490."}