{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RMMGUOW6B2RDHVCMREU7OGCFLK","short_pith_number":"pith:RMMGUOW6","schema_version":"1.0","canonical_sha256":"8b186a3ade0ea233d44c8929f718455aa3f6182dc3236064211fbfba2f2f006b","source":{"kind":"arxiv","id":"2411.02391","version":2},"attestation_state":"computed","paper":{"title":"Attacking Vision-Language Computer Agents via Pop-ups","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Diyi Yang, Tao Yu, Yanzhe Zhang","submitted_at":"2024-11-04T18:56:42Z","abstract_excerpt":"Autonomous agents powered by large vision and language models (VLM) have demonstrated significant potential in completing daily computer tasks, such as browsing the web to book travel and operating desktop software, which requires agents to understand these interfaces. Despite such visual inputs becoming more integrated into agentic applications, what types of risks and attacks exist around them still remain unclear. In this work, we demonstrate that VLM agents can be easily attacked by a set of carefully designed adversarial pop-ups, which human users would typically recognize and ignore. Thi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.02391","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-04T18:56:42Z","cross_cats_sorted":[],"title_canon_sha256":"145061438d4647ad1b3bb84d2286dfa07cdcf5b1b7e8b266a8d9a02231aa0a0c","abstract_canon_sha256":"6ff6b31274ad8a05b6515828015d1d515c6a27d7c2278e27f62690a0c68d684c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:03.707625Z","signature_b64":"N8ShcaEJHwTUIbbRIUs9cdduz9nHO2MucKBtd5hwdJmq5YwudONMtf6nfX5dJOpgHk6SXKn9bnDVrLi05MSABA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b186a3ade0ea233d44c8929f718455aa3f6182dc3236064211fbfba2f2f006b","last_reissued_at":"2026-07-05T11:09:03.707172Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:03.707172Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Attacking Vision-Language Computer Agents via Pop-ups","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Diyi Yang, Tao Yu, Yanzhe Zhang","submitted_at":"2024-11-04T18:56:42Z","abstract_excerpt":"Autonomous agents powered by large vision and language models (VLM) have demonstrated significant potential in completing daily computer tasks, such as browsing the web to book travel and operating desktop software, which requires agents to understand these interfaces. Despite such visual inputs becoming more integrated into agentic applications, what types of risks and attacks exist around them still remain unclear. In this work, we demonstrate that VLM agents can be easily attacked by a set of carefully designed adversarial pop-ups, which human users would typically recognize and ignore. Thi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.02391","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.02391/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.02391","created_at":"2026-07-05T11:09:03.707231+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.02391v2","created_at":"2026-07-05T11:09:03.707231+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.02391","created_at":"2026-07-05T11:09:03.707231+00:00"},{"alias_kind":"pith_short_12","alias_value":"RMMGUOW6B2RD","created_at":"2026-07-05T11:09:03.707231+00:00"},{"alias_kind":"pith_short_16","alias_value":"RMMGUOW6B2RDHVCM","created_at":"2026-07-05T11:09:03.707231+00:00"},{"alias_kind":"pith_short_8","alias_value":"RMMGUOW6","created_at":"2026-07-05T11:09:03.707231+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08147","citing_title":"Prismata: Confining Cross-Site Prompt Injection in Web Agents","ref_index":99,"is_internal_anchor":true},{"citing_arxiv_id":"2606.05233","citing_title":"Domain-Conditioned Safety in Frontier Computer-Using Agents: A 793-Episode Browser Benchmark, a Coding-Domain Cross-Reference, and a Reproducibility Audit of Recent Red-Teaming","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15030","citing_title":"WARD: Adversarially Robust Defense of Web Agents Against Prompt Injections","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2606.14027","citing_title":"Same-Origin Policy for Agentic Browsers","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06708","citing_title":"Signal-Driven Observation for Long-Horizon Web Agents","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2504.11703","citing_title":"Progent: Securing AI Agents with Privilege Control","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2506.23978","citing_title":"LLM Agents Are the Antidote to Walled Gardens","ref_index":106,"is_internal_anchor":false},{"citing_arxiv_id":"2507.04227","citing_title":"Mobile GUI Agents under Real-world Threats: Are We There Yet?","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2507.10610","citing_title":"LaSM: Layer-wise Scaling Mechanism for Defending Pop-up Attack on GUI Agents","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2510.10073","citing_title":"SecureWebArena: A Holistic Security Evaluation Benchmark for LVLM-based Web Agents","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2510.23883","citing_title":"Agentic AI Security: Threats, Defenses, Evaluation, and Open Challenges","ref_index":73,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RMMGUOW6B2RDHVCMREU7OGCFLK","json":"https://pith.science/pith/RMMGUOW6B2RDHVCMREU7OGCFLK.json","graph_json":"https://pith.science/api/pith-number/RMMGUOW6B2RDHVCMREU7OGCFLK/graph.json","events_json":"https://pith.science/api/pith-number/RMMGUOW6B2RDHVCMREU7OGCFLK/events.json","paper":"https://pith.science/paper/RMMGUOW6"},"agent_actions":{"view_html":"https://pith.science/pith/RMMGUOW6B2RDHVCMREU7OGCFLK","download_json":"https://pith.science/pith/RMMGUOW6B2RDHVCMREU7OGCFLK.json","view_paper":"https://pith.science/paper/RMMGUOW6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.02391&json=true","fetch_graph":"https://pith.science/api/pith-number/RMMGUOW6B2RDHVCMREU7OGCFLK/graph.json","fetch_events":"https://pith.science/api/pith-number/RMMGUOW6B2RDHVCMREU7OGCFLK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RMMGUOW6B2RDHVCMREU7OGCFLK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RMMGUOW6B2RDHVCMREU7OGCFLK/action/storage_attestation","attest_author":"https://pith.science/pith/RMMGUOW6B2RDHVCMREU7OGCFLK/action/author_attestation","sign_citation":"https://pith.science/pith/RMMGUOW6B2RDHVCMREU7OGCFLK/action/citation_signature","submit_replication":"https://pith.science/pith/RMMGUOW6B2RDHVCMREU7OGCFLK/action/replication_record"}},"created_at":"2026-07-05T11:09:03.707231+00:00","updated_at":"2026-07-05T11:09:03.707231+00:00"}