{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:R6CB7RKXK7GQL6K2FKJ5WBUAHY","short_pith_number":"pith:R6CB7RKX","schema_version":"1.0","canonical_sha256":"8f841fc55757cd05f95a2a93db06803e0f4e936ea1f4eb98ba228353a3758d6d","source":{"kind":"arxiv","id":"2402.10340","version":5},"attestation_state":"computed","paper":{"title":"On the Vulnerability of LLM/VLM-Controlled Robotics","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Amrit Singh Bedi, Brian M. Sadler, Dinesh Manocha, Fuxiao Liu, Jing Liang, Ruiqi Xian, Souradip Chakraborty, Tianrui Guan, Xiyang Wu","submitted_at":"2024-02-15T22:01:45Z","abstract_excerpt":"In this work, we highlight vulnerabilities in robotic systems integrating large language models (LLMs) and vision-language models (VLMs) due to input modality sensitivities. While LLM/VLM-controlled robots show impressive performance across various tasks, their reliability under slight input variations remains underexplored yet critical. These models are highly sensitive to instruction or perceptual input changes, which can trigger misalignment issues, leading to execution failures with severe real-world consequences. To study this issue, we analyze the misalignment-induced vulnerabilities wit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.10340","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.RO","submitted_at":"2024-02-15T22:01:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2becf61e38c068ae21b05d30d64912081c85cff407052469e496b0b4d97432d0","abstract_canon_sha256":"a329dff822711beceb5469e059bf67f55353823cf7891ced3f1ba6a5462fac2d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:26:01.253661Z","signature_b64":"CI7w6S4vMzqQCKw7ICO+sVEF7CPvAsUrhtgBNUH8WbGcK1noG2M053KwiMOGojWQfYC3BIhdJJ9TfWZ1sLpGBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8f841fc55757cd05f95a2a93db06803e0f4e936ea1f4eb98ba228353a3758d6d","last_reissued_at":"2026-07-05T10:26:01.252742Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:26:01.252742Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Vulnerability of LLM/VLM-Controlled Robotics","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Amrit Singh Bedi, Brian M. Sadler, Dinesh Manocha, Fuxiao Liu, Jing Liang, Ruiqi Xian, Souradip Chakraborty, Tianrui Guan, Xiyang Wu","submitted_at":"2024-02-15T22:01:45Z","abstract_excerpt":"In this work, we highlight vulnerabilities in robotic systems integrating large language models (LLMs) and vision-language models (VLMs) due to input modality sensitivities. While LLM/VLM-controlled robots show impressive performance across various tasks, their reliability under slight input variations remains underexplored yet critical. These models are highly sensitive to instruction or perceptual input changes, which can trigger misalignment issues, leading to execution failures with severe real-world consequences. To study this issue, we analyze the misalignment-induced vulnerabilities wit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.10340","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.10340/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.10340","created_at":"2026-07-05T10:26:01.252870+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.10340v5","created_at":"2026-07-05T10:26:01.252870+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.10340","created_at":"2026-07-05T10:26:01.252870+00:00"},{"alias_kind":"pith_short_12","alias_value":"R6CB7RKXK7GQ","created_at":"2026-07-05T10:26:01.252870+00:00"},{"alias_kind":"pith_short_16","alias_value":"R6CB7RKXK7GQL6K2","created_at":"2026-07-05T10:26:01.252870+00:00"},{"alias_kind":"pith_short_8","alias_value":"R6CB7RKX","created_at":"2026-07-05T10:26:01.252870+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.03433","citing_title":"When control meets large language models: From words to dynamics","ref_index":214,"is_internal_anchor":false},{"citing_arxiv_id":"2512.21815","citing_title":"High-Entropy Tokens as Multimodal Failure Points in Vision-Language Models","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R6CB7RKXK7GQL6K2FKJ5WBUAHY","json":"https://pith.science/pith/R6CB7RKXK7GQL6K2FKJ5WBUAHY.json","graph_json":"https://pith.science/api/pith-number/R6CB7RKXK7GQL6K2FKJ5WBUAHY/graph.json","events_json":"https://pith.science/api/pith-number/R6CB7RKXK7GQL6K2FKJ5WBUAHY/events.json","paper":"https://pith.science/paper/R6CB7RKX"},"agent_actions":{"view_html":"https://pith.science/pith/R6CB7RKXK7GQL6K2FKJ5WBUAHY","download_json":"https://pith.science/pith/R6CB7RKXK7GQL6K2FKJ5WBUAHY.json","view_paper":"https://pith.science/paper/R6CB7RKX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.10340&json=true","fetch_graph":"https://pith.science/api/pith-number/R6CB7RKXK7GQL6K2FKJ5WBUAHY/graph.json","fetch_events":"https://pith.science/api/pith-number/R6CB7RKXK7GQL6K2FKJ5WBUAHY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R6CB7RKXK7GQL6K2FKJ5WBUAHY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R6CB7RKXK7GQL6K2FKJ5WBUAHY/action/storage_attestation","attest_author":"https://pith.science/pith/R6CB7RKXK7GQL6K2FKJ5WBUAHY/action/author_attestation","sign_citation":"https://pith.science/pith/R6CB7RKXK7GQL6K2FKJ5WBUAHY/action/citation_signature","submit_replication":"https://pith.science/pith/R6CB7RKXK7GQL6K2FKJ5WBUAHY/action/replication_record"}},"created_at":"2026-07-05T10:26:01.252870+00:00","updated_at":"2026-07-05T10:26:01.252870+00:00"}