{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:52JP5X3OGDQNHST7UVIVX2J7K6","short_pith_number":"pith:52JP5X3O","schema_version":"1.0","canonical_sha256":"ee92fedf6e30e0d3ca7fa5515be93f57b98e0f85ac8a6d2874967e67e0836f31","source":{"kind":"arxiv","id":"2412.00114","version":2},"attestation_state":"computed","paper":{"title":"SceneTAP: Scene-Coherent Typographic Adversarial Planner against Vision-Language Models in Real-World Environments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Di Lin, Ivor Tsang, Jie Zhang, Qing Guo, Tianwei Zhang, Yang Liu, Yue Cao, Yun Xing","submitted_at":"2024-11-28T05:55:13Z","abstract_excerpt":"Large vision-language models (LVLMs) have shown remarkable capabilities in interpreting visual content. While existing works demonstrate these models' vulnerability to deliberately placed adversarial texts, such texts are often easily identifiable as anomalous. In this paper, we present the first approach to generate scene-coherent typographic adversarial attacks that mislead advanced LVLMs while maintaining visual naturalness through the capability of the LLM-based agent. Our approach addresses three critical questions: what adversarial text to generate, where to place it within the scene, an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.00114","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-28T05:55:13Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b66366ff2b825bae449715003509998bfc473888e9421de2402868264f2f76c7","abstract_canon_sha256":"24c93d0ef7f5a5d73a67a5efc03012b3f879d2250332b6777a724c682a9dead6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:45:57.720648Z","signature_b64":"vX9DZxCAAbltM1OjvzGOypks2IK8JDd78oTREk6791tV04PdSCSu62iRz+VLeue5xYuL6Xo00BFw2JkXMbsnAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ee92fedf6e30e0d3ca7fa5515be93f57b98e0f85ac8a6d2874967e67e0836f31","last_reissued_at":"2026-07-05T10:45:57.720176Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:45:57.720176Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SceneTAP: Scene-Coherent Typographic Adversarial Planner against Vision-Language Models in Real-World Environments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Di Lin, Ivor Tsang, Jie Zhang, Qing Guo, Tianwei Zhang, Yang Liu, Yue Cao, Yun Xing","submitted_at":"2024-11-28T05:55:13Z","abstract_excerpt":"Large vision-language models (LVLMs) have shown remarkable capabilities in interpreting visual content. While existing works demonstrate these models' vulnerability to deliberately placed adversarial texts, such texts are often easily identifiable as anomalous. In this paper, we present the first approach to generate scene-coherent typographic adversarial attacks that mislead advanced LVLMs while maintaining visual naturalness through the capability of the LLM-based agent. Our approach addresses three critical questions: what adversarial text to generate, where to place it within the scene, an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.00114","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.00114/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.00114","created_at":"2026-07-05T10:45:57.720231+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.00114v2","created_at":"2026-07-05T10:45:57.720231+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.00114","created_at":"2026-07-05T10:45:57.720231+00:00"},{"alias_kind":"pith_short_12","alias_value":"52JP5X3OGDQN","created_at":"2026-07-05T10:45:57.720231+00:00"},{"alias_kind":"pith_short_16","alias_value":"52JP5X3OGDQNHST7","created_at":"2026-07-05T10:45:57.720231+00:00"},{"alias_kind":"pith_short_8","alias_value":"52JP5X3O","created_at":"2026-07-05T10:45:57.720231+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18593","citing_title":"Not What You Asked For: Typographic Attacks in Household Robot Manipulation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25102","citing_title":"One Perturbation, Two Failure Modes: Probing VLM Safety via Embedding-Guided Typographic Perturbations","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/52JP5X3OGDQNHST7UVIVX2J7K6","json":"https://pith.science/pith/52JP5X3OGDQNHST7UVIVX2J7K6.json","graph_json":"https://pith.science/api/pith-number/52JP5X3OGDQNHST7UVIVX2J7K6/graph.json","events_json":"https://pith.science/api/pith-number/52JP5X3OGDQNHST7UVIVX2J7K6/events.json","paper":"https://pith.science/paper/52JP5X3O"},"agent_actions":{"view_html":"https://pith.science/pith/52JP5X3OGDQNHST7UVIVX2J7K6","download_json":"https://pith.science/pith/52JP5X3OGDQNHST7UVIVX2J7K6.json","view_paper":"https://pith.science/paper/52JP5X3O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.00114&json=true","fetch_graph":"https://pith.science/api/pith-number/52JP5X3OGDQNHST7UVIVX2J7K6/graph.json","fetch_events":"https://pith.science/api/pith-number/52JP5X3OGDQNHST7UVIVX2J7K6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/52JP5X3OGDQNHST7UVIVX2J7K6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/52JP5X3OGDQNHST7UVIVX2J7K6/action/storage_attestation","attest_author":"https://pith.science/pith/52JP5X3OGDQNHST7UVIVX2J7K6/action/author_attestation","sign_citation":"https://pith.science/pith/52JP5X3OGDQNHST7UVIVX2J7K6/action/citation_signature","submit_replication":"https://pith.science/pith/52JP5X3OGDQNHST7UVIVX2J7K6/action/replication_record"}},"created_at":"2026-07-05T10:45:57.720231+00:00","updated_at":"2026-07-05T10:45:57.720231+00:00"}