{"as_of":"2026-08-06T20:21:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:399a0b43ee10371c0292d98ae597dcd0caec7f72239b31ac5bcaf02827506e18","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":10,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":10,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":10,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":10,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:09:24.061030Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T19:58:53.646539Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.00626","last_updated":"2025-02-13T03:11:20Z","snapshot_observed_at":"2026-07-06T17:23:43.676786Z","submitted_at":"2024-02-01T14:41:20Z","title":"Vision-LLMs Can Fool Themselves with Self-Generated Typographic Attacks","version":3},"cited_work":{"arxiv_id":"2402.00626","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.00626","snapshot_observed_at":"2026-07-03T19:58:53.646539Z","title":"Vision-llms can fool themselves with self-generated typographic attacks","venue":null,"work_id":"6a2505a9-5b89-4cb6-ad94-b7631346995c","year":2024},"citing_paper":{"arxiv_id":"2502.05206","last_updated":"2026-04-14T16:10:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-02T05:14:22Z","title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","version":6},"reference_index":294,"source":"pdf_text","source_observed_at":"2026-05-23T04:39:04.591722Z"},"links":{"cited_paper":"/paper/2402.00626","citing_paper":"/paper/2502.05206"},"observation_digest":"sha256:9381f8b6430eff5e902635865ed611523ac9ddad8d657190b2a607f5e754cae5","observation_id":"d902c91b-1bc2-4c0f-bf22-4402e9b5bbd7","resolution":{"observed_at":"2026-05-23T04:42:33.969912Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.00626","last_updated":"2025-02-13T03:11:20Z","snapshot_observed_at":"2026-07-06T17:23:43.676786Z","submitted_at":"2024-02-01T14:41:20Z","title":"Vision-LLMs Can Fool Themselves with Self-Generated Typographic Attacks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.00626","snapshot_observed_at":"2026-08-06T17:09:24.061030Z","title":"Vision-llms can fool themselves with self-generated typographic attacks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11968","last_updated":"2025-07-16T07:02:15Z","snapshot_observed_at":"2026-08-06T16:54:49.708049Z","submitted_at":"2025-07-16T07:02:15Z","title":"Watch, Listen, Understand, Mislead: Tri-modal Adversarial Attacks on Short Videos for Content Appropriateness Evaluation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T17:09:24.061030Z"},"links":{"cited_paper":"/paper/2402.00626","citing_paper":"/paper/2507.11968"},"observation_digest":"sha256:a747c7331e05aec440532db7e7f343145861eb6356874bad002c1070369f1a24","observation_id":"fac8730f-908b-48c4-93c5-9c6ed93363b0","resolution":{"observed_at":"2026-08-06T17:09:24.061030Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.00626","last_updated":"2025-02-13T03:11:20Z","snapshot_observed_at":"2026-07-06T17:23:43.676786Z","submitted_at":"2024-02-01T14:41:20Z","title":"Vision-LLMs Can Fool Themselves with Self-Generated Typographic Attacks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.00626","snapshot_observed_at":"2026-08-05T20:29:05.912718Z","title":"Qraitem et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.10955","last_updated":"2025-08-14T07:25:45Z","snapshot_observed_at":"2026-08-06T06:18:15.087722Z","submitted_at":"2025-08-14T07:25:45Z","title":"Empowering Multimodal LLMs with External Tools: A Comprehensive Survey","version":1},"reference_index":226,"source":"arxiv_source","source_observed_at":"2026-08-05T20:29:05.912718Z"},"links":{"cited_paper":"/paper/2402.00626","citing_paper":"/paper/2508.10955"},"observation_digest":"sha256:9a4b5887b0f528472cf7fe49ccce4c521fddcf8ca2c60cec0ef7d1a2ea29b76c","observation_id":"b64cd411-d2df-484b-b9a4-a53bd96f9064","resolution":{"observed_at":"2026-08-05T20:29:05.912718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.00626","last_updated":"2025-02-13T03:11:20Z","snapshot_observed_at":"2026-07-06T17:23:43.676786Z","submitted_at":"2024-02-01T14:41:20Z","title":"Vision-LLMs Can Fool Themselves with Self-Generated Typographic Attacks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.00626","snapshot_observed_at":"2026-08-05T16:00:48.848984Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.19445","last_updated":"2026-06-16T17:34:06Z","snapshot_observed_at":"2026-08-05T21:13:40.763863Z","submitted_at":"2025-08-26T21:36:45Z","title":"On Surjectivity of Neural Networks: Can you elicit any behavior from your model?","version":3},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-05T16:00:48.848984Z"},"links":{"cited_paper":"/paper/2402.00626","citing_paper":"/paper/2508.19445"},"observation_digest":"sha256:33980a1293317b48350ec3305fdb9df41b4eb006932f74548dd984df05785f0a","observation_id":"f6be0366-bb58-4df4-bfc0-216ecec2d5fe","resolution":{"observed_at":"2026-08-05T16:00:48.848984Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.00626","last_updated":"2025-02-13T03:11:20Z","snapshot_observed_at":"2026-07-06T17:23:43.676786Z","submitted_at":"2024-02-01T14:41:20Z","title":"Vision-LLMs Can Fool Themselves with Self-Generated Typographic Attacks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.00626","snapshot_observed_at":"2026-08-04T09:02:39.623038Z","title":"Vision-llms can fool themselves with self-generated typographic attacks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.17759","last_updated":"2026-05-26T07:09:27Z","snapshot_observed_at":"2026-08-04T09:02:34.047795Z","submitted_at":"2025-10-20T17:12:10Z","title":"VERA-V: Variational Inference Framework for Jailbreaking Vision-Language Models","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-04T09:02:39.623038Z"},"links":{"cited_paper":"/paper/2402.00626","citing_paper":"/paper/2510.17759"},"observation_digest":"sha256:030f398c58ea183508486fb0c917f17a32ae4342127d7eca5132866975f028ae","observation_id":"ae799f24-33b2-466b-b074-5fb104fdb242","resolution":{"observed_at":"2026-08-04T09:02:39.623038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.00626","last_updated":"2025-02-13T03:11:20Z","snapshot_observed_at":"2026-07-06T17:23:43.676786Z","submitted_at":"2024-02-01T14:41:20Z","title":"Vision-LLMs Can Fool Themselves with Self-Generated Typographic Attacks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.00626","snapshot_observed_at":"2026-08-03T17:32:15.720550Z","title":"Vision-llms can fool themselves with self-generated typographic attacks.arXiv preprint arXiv:2402.00626, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2512.11899","last_updated":"2026-07-21T16:21:35Z","snapshot_observed_at":"2026-08-03T23:38:42.360154Z","submitted_at":"2025-12-10T08:34:28Z","title":"Read or Ignore? A Unified Benchmark for Typographic-Attack Robustness and Text Recognition in Vision-Language Models","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-03T17:32:15.720550Z"},"links":{"cited_paper":"/paper/2402.00626","citing_paper":"/paper/2512.11899"},"observation_digest":"sha256:84e6e9e446cb0ceaf625a8c98d6c2fce952707bed52309d99fcf20707b28e63a","observation_id":"bae15cad-b497-4d7f-8413-2808442bd699","resolution":{"observed_at":"2026-08-03T17:32:15.720550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.00626","last_updated":"2025-02-13T03:11:20Z","snapshot_observed_at":"2026-07-06T17:23:43.676786Z","submitted_at":"2024-02-01T14:41:20Z","title":"Vision-LLMs Can Fool Themselves with Self-Generated Typographic Attacks","version":3},"cited_work":{"arxiv_id":"2402.00626","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.00626","snapshot_observed_at":"2026-07-03T19:58:53.646539Z","title":"Vision-llms can fool themselves with self-generated typographic attacks","venue":null,"work_id":"6a2505a9-5b89-4cb6-ad94-b7631346995c","year":2024},"citing_paper":{"arxiv_id":"2604.03995","last_updated":"2026-04-05T06:32:08Z","snapshot_observed_at":"2026-07-06T22:53:01.835843Z","submitted_at":"2026-04-05T06:32:08Z","title":"A Systematic Study of Cross-Modal Typographic Attacks on Audio-Visual Reasoning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-13T17:34:10.089555Z"},"links":{"cited_paper":"/paper/2402.00626","citing_paper":"/paper/2604.03995"},"observation_digest":"sha256:24ea737bbcc7db04803c23368d7c39d6e414ca05ace9233dbac29663f380d139","observation_id":"bc7b464c-7e41-40f4-9d14-a4616f6bb9a7","resolution":{"observed_at":"2026-05-13T17:38:02.892843Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.00626","last_updated":"2025-02-13T03:11:20Z","snapshot_observed_at":"2026-07-06T17:23:43.676786Z","submitted_at":"2024-02-01T14:41:20Z","title":"Vision-LLMs Can Fool Themselves with Self-Generated Typographic Attacks","version":3},"cited_work":{"arxiv_id":"2402.00626","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.00626","snapshot_observed_at":"2026-07-03T19:58:53.646539Z","title":"Vision-llms can fool themselves with self-generated typographic attacks","venue":null,"work_id":"6a2505a9-5b89-4cb6-ad94-b7631346995c","year":2024},"citing_paper":{"arxiv_id":"2605.11716","last_updated":"2026-05-12T08:05:10Z","snapshot_observed_at":"2026-07-06T23:23:30.499023Z","submitted_at":"2026-05-12T08:05:10Z","title":"SafeSteer: A Decoding-level Defense Mechanism for Multimodal Large Language Models","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-05-13T06:56:10.053418Z"},"links":{"cited_paper":"/paper/2402.00626","citing_paper":"/paper/2605.11716"},"observation_digest":"sha256:52032026857b17c240b5d8dcdc9d8a48a6dbedd6f159f0f62b72f6eda99b62af","observation_id":"4e6ee125-5564-4257-9c71-d84db55ad4d8","resolution":{"observed_at":"2026-05-13T06:57:27.669590Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.00626","last_updated":"2025-02-13T03:11:20Z","snapshot_observed_at":"2026-07-06T17:23:43.676786Z","submitted_at":"2024-02-01T14:41:20Z","title":"Vision-LLMs Can Fool Themselves with Self-Generated Typographic Attacks","version":3},"cited_work":{"arxiv_id":"2402.00626","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.00626","snapshot_observed_at":"2026-07-03T19:58:53.646539Z","title":"Vision-llms can fool themselves with self-generated typographic attacks","venue":null,"work_id":"6a2505a9-5b89-4cb6-ad94-b7631346995c","year":2024},"citing_paper":{"arxiv_id":"2607.01518","last_updated":"2026-07-01T22:31:10Z","snapshot_observed_at":"2026-08-06T19:26:37.371608Z","submitted_at":"2026-07-01T22:31:10Z","title":"Overthink-Triggered Slowdown Attacks on LVLM-Based Robotic Systems","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-03T19:52:11.018335Z"},"links":{"cited_paper":"/paper/2402.00626","citing_paper":"/paper/2607.01518"},"observation_digest":"sha256:a3982b8c1d335061789a1a3e756ed83a60c79c9951b03f2a5e5142f9a515f129","observation_id":"34b715b1-8df7-46f1-bc91-fa514fce4b35","resolution":{"observed_at":"2026-07-03T19:58:53.648368Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.00626","last_updated":"2025-02-13T03:11:20Z","snapshot_observed_at":"2026-07-06T17:23:43.676786Z","submitted_at":"2024-02-01T14:41:20Z","title":"Vision-LLMs Can Fool Themselves with Self-Generated Typographic Attacks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.00626","snapshot_observed_at":"2026-07-14T13:02:42.673767Z","title":"2024.Vision-llms can fool themselves with self-generated typographic attacks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.10269","last_updated":"2026-07-11T11:56:05Z","snapshot_observed_at":"2026-08-06T13:53:30.569509Z","submitted_at":"2026-07-11T11:56:05Z","title":"Devil in the Lens: Analyzing and Defending Physical Prompt Injection Against Vision-Language Models on Wearable Devices","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-14T13:02:42.673767Z"},"links":{"cited_paper":"/paper/2402.00626","citing_paper":"/paper/2607.10269"},"observation_digest":"sha256:c69e9db99581b5f2e0055b3da6d8dfd29beaf2705177b4ecf0bc223bf114bf6a","observation_id":"88d15d79-a94a-4377-8555-f4b57c9b6586","resolution":{"observed_at":"2026-07-14T13:02:42.673767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2402.00626/citation-record","integrity":"/paper/2402.00626/integrity","json":"/paper/2402.00626/citation-record.json","paper":"/paper/2402.00626"},"outbound":[],"paper":{"arxiv_id":"2402.00626","last_updated":"2025-02-13T03:11:20Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T17:23:43.676786Z","submitted_at":"2024-02-01T14:41:20Z","title":"Vision-LLMs Can Fool Themselves with Self-Generated Typographic Attacks"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 10 inbound Pith citation observations for arXiv:2402.00626."}