{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FXDMQUYP4JYX2JWGFDSSE6EDLZ","short_pith_number":"pith:FXDMQUYP","schema_version":"1.0","canonical_sha256":"2dc6c8530fe2717d26c628e52278835e7f3370c84b21811d77125d7680a7ad8c","source":{"kind":"arxiv","id":"2411.13587","version":4},"attestation_state":"computed","paper":{"title":"Exploring the Adversarial Vulnerabilities of Vision-Language-Action Models in Robotics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Cheng Han, Dongfang Liu, James Chenhao Liang, Jiebo Luo, Luna Xinyu Zhang, Qifan Wang, Ruixiang Tang, Taowen Wang, Wenhao Yang","submitted_at":"2024-11-18T01:52:20Z","abstract_excerpt":"Recently in robotics, Vision-Language-Action (VLA) models have emerged as a transformative approach, enabling robots to execute complex tasks by integrating visual and linguistic inputs within an end-to-end learning framework. Despite their significant capabilities, VLA models introduce new attack surfaces. This paper systematically evaluates their robustness. Recognizing the unique demands of robotic execution, our attack objectives target the inherent spatial and functional characteristics of robotic systems. In particular, we introduce two untargeted attack objectives that leverage spatial "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.13587","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-11-18T01:52:20Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7e405cf449803e612ca68ca92b207ee31636e98faeeae4494026a3aa91ee43bc","abstract_canon_sha256":"a5b7104796678d3eb2b1c3f1cc8c8fccbefc9c8c58c70792ae4b43053f63e779"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:46:37.251213Z","signature_b64":"JmZ3NW4WBwGYQDNWYFMMFj9CHX4galVTWRqqNE2FIBsoxf57qxphXUafCvgBDXmcBbhjMDU8YawwWN+CxagDAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2dc6c8530fe2717d26c628e52278835e7f3370c84b21811d77125d7680a7ad8c","last_reissued_at":"2026-07-05T11:46:37.250672Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:46:37.250672Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring the Adversarial Vulnerabilities of Vision-Language-Action Models in Robotics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Cheng Han, Dongfang Liu, James Chenhao Liang, Jiebo Luo, Luna Xinyu Zhang, Qifan Wang, Ruixiang Tang, Taowen Wang, Wenhao Yang","submitted_at":"2024-11-18T01:52:20Z","abstract_excerpt":"Recently in robotics, Vision-Language-Action (VLA) models have emerged as a transformative approach, enabling robots to execute complex tasks by integrating visual and linguistic inputs within an end-to-end learning framework. Despite their significant capabilities, VLA models introduce new attack surfaces. This paper systematically evaluates their robustness. Recognizing the unique demands of robotic execution, our attack objectives target the inherent spatial and functional characteristics of robotic systems. In particular, we introduce two untargeted attack objectives that leverage spatial "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.13587","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.13587/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.13587","created_at":"2026-07-05T11:46:37.250733+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.13587v4","created_at":"2026-07-05T11:46:37.250733+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.13587","created_at":"2026-07-05T11:46:37.250733+00:00"},{"alias_kind":"pith_short_12","alias_value":"FXDMQUYP4JYX","created_at":"2026-07-05T11:46:37.250733+00:00"},{"alias_kind":"pith_short_16","alias_value":"FXDMQUYP4JYX2JWG","created_at":"2026-07-05T11:46:37.250733+00:00"},{"alias_kind":"pith_short_8","alias_value":"FXDMQUYP","created_at":"2026-07-05T11:46:37.250733+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18632","citing_title":"ROBOSHACKLES: A Safety Dataset for Human-Injury Prevention in Embodied Foundation Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12978","citing_title":"Trajectory-Level Redirection Attacks on Vision-Language-Action Models","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28083","citing_title":"VLA-Hijack: A Transferable Patch Attack against Vision-Language-Action Models via Visual Proprioception Hijacking","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10371","citing_title":"Test-time Adversarial Takeover: A Real-time Hijacking Interface against Robotic Diffusion Policies","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2510.03827","citing_title":"LIBERO-PRO: Towards Robust and Fair Evaluation of Vision-Language-Action Models Beyond Memorization","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FXDMQUYP4JYX2JWGFDSSE6EDLZ","json":"https://pith.science/pith/FXDMQUYP4JYX2JWGFDSSE6EDLZ.json","graph_json":"https://pith.science/api/pith-number/FXDMQUYP4JYX2JWGFDSSE6EDLZ/graph.json","events_json":"https://pith.science/api/pith-number/FXDMQUYP4JYX2JWGFDSSE6EDLZ/events.json","paper":"https://pith.science/paper/FXDMQUYP"},"agent_actions":{"view_html":"https://pith.science/pith/FXDMQUYP4JYX2JWGFDSSE6EDLZ","download_json":"https://pith.science/pith/FXDMQUYP4JYX2JWGFDSSE6EDLZ.json","view_paper":"https://pith.science/paper/FXDMQUYP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.13587&json=true","fetch_graph":"https://pith.science/api/pith-number/FXDMQUYP4JYX2JWGFDSSE6EDLZ/graph.json","fetch_events":"https://pith.science/api/pith-number/FXDMQUYP4JYX2JWGFDSSE6EDLZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FXDMQUYP4JYX2JWGFDSSE6EDLZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FXDMQUYP4JYX2JWGFDSSE6EDLZ/action/storage_attestation","attest_author":"https://pith.science/pith/FXDMQUYP4JYX2JWGFDSSE6EDLZ/action/author_attestation","sign_citation":"https://pith.science/pith/FXDMQUYP4JYX2JWGFDSSE6EDLZ/action/citation_signature","submit_replication":"https://pith.science/pith/FXDMQUYP4JYX2JWGFDSSE6EDLZ/action/replication_record"}},"created_at":"2026-07-05T11:46:37.250733+00:00","updated_at":"2026-07-05T11:46:37.250733+00:00"}