{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GP3UHDERY3BNWBDAWH3R2NDNHM","short_pith_number":"pith:GP3UHDER","schema_version":"1.0","canonical_sha256":"33f7438c91c6c2db0460b1f71d346d3b38fe1893132ded41c3e48c0566ccd7f0","source":{"kind":"arxiv","id":"2407.09829","version":1},"attestation_state":"computed","paper":{"title":"VLMPC: Vision-Language Model Predictive Control for Robotic Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Donghui Mao, Jiaming Chen, Ran Song, Wei Zhang, Wentao Zhao, Ziyu Meng","submitted_at":"2024-07-13T09:42:02Z","abstract_excerpt":"Although Model Predictive Control (MPC) can effectively predict the future states of a system and thus is widely used in robotic manipulation tasks, it does not have the capability of environmental perception, leading to the failure in some complex scenarios. To address this issue, we introduce Vision-Language Model Predictive Control (VLMPC), a robotic manipulation framework which takes advantage of the powerful perception capability of vision language model (VLM) and integrates it with MPC. Specifically, we propose a conditional action sampling module which takes as input a goal image or a l"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.09829","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-07-13T09:42:02Z","cross_cats_sorted":[],"title_canon_sha256":"9c0f062cbace73b61d1a7e0eea06748393a77f96b8d21da21512344488e40b33","abstract_canon_sha256":"ac7dc2aab3eea8c9935990ded18a57380ea4a0feea8dbb9534353108e94e0774"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:43:35.255438Z","signature_b64":"KCVeEvLIbGBUq/utLWRao9JuphtJ6FexPW0fSq2PWlbZxy1S+lOSlKAJN9edkzYGpALk79EWf8j/GQA3Ey77AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"33f7438c91c6c2db0460b1f71d346d3b38fe1893132ded41c3e48c0566ccd7f0","last_reissued_at":"2026-07-05T08:43:35.254942Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:43:35.254942Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VLMPC: Vision-Language Model Predictive Control for Robotic Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Donghui Mao, Jiaming Chen, Ran Song, Wei Zhang, Wentao Zhao, Ziyu Meng","submitted_at":"2024-07-13T09:42:02Z","abstract_excerpt":"Although Model Predictive Control (MPC) can effectively predict the future states of a system and thus is widely used in robotic manipulation tasks, it does not have the capability of environmental perception, leading to the failure in some complex scenarios. To address this issue, we introduce Vision-Language Model Predictive Control (VLMPC), a robotic manipulation framework which takes advantage of the powerful perception capability of vision language model (VLM) and integrates it with MPC. Specifically, we propose a conditional action sampling module which takes as input a goal image or a l"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.09829","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.09829/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.09829","created_at":"2026-07-05T08:43:35.255004+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.09829v1","created_at":"2026-07-05T08:43:35.255004+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.09829","created_at":"2026-07-05T08:43:35.255004+00:00"},{"alias_kind":"pith_short_12","alias_value":"GP3UHDERY3BN","created_at":"2026-07-05T08:43:35.255004+00:00"},{"alias_kind":"pith_short_16","alias_value":"GP3UHDERY3BNWBDA","created_at":"2026-07-05T08:43:35.255004+00:00"},{"alias_kind":"pith_short_8","alias_value":"GP3UHDER","created_at":"2026-07-05T08:43:35.255004+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21572","citing_title":"Robot Critics that Sweat the Small Stuff","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29350","citing_title":"Fast Enough to Act: Spatio-Temporal Visual Token Merging for Low-Latency Robotic VLMs and VLAs","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11751","citing_title":"Grounded World Model for Semantically Generalizable Planning","ref_index":66,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GP3UHDERY3BNWBDAWH3R2NDNHM","json":"https://pith.science/pith/GP3UHDERY3BNWBDAWH3R2NDNHM.json","graph_json":"https://pith.science/api/pith-number/GP3UHDERY3BNWBDAWH3R2NDNHM/graph.json","events_json":"https://pith.science/api/pith-number/GP3UHDERY3BNWBDAWH3R2NDNHM/events.json","paper":"https://pith.science/paper/GP3UHDER"},"agent_actions":{"view_html":"https://pith.science/pith/GP3UHDERY3BNWBDAWH3R2NDNHM","download_json":"https://pith.science/pith/GP3UHDERY3BNWBDAWH3R2NDNHM.json","view_paper":"https://pith.science/paper/GP3UHDER","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.09829&json=true","fetch_graph":"https://pith.science/api/pith-number/GP3UHDERY3BNWBDAWH3R2NDNHM/graph.json","fetch_events":"https://pith.science/api/pith-number/GP3UHDERY3BNWBDAWH3R2NDNHM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GP3UHDERY3BNWBDAWH3R2NDNHM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GP3UHDERY3BNWBDAWH3R2NDNHM/action/storage_attestation","attest_author":"https://pith.science/pith/GP3UHDERY3BNWBDAWH3R2NDNHM/action/author_attestation","sign_citation":"https://pith.science/pith/GP3UHDERY3BNWBDAWH3R2NDNHM/action/citation_signature","submit_replication":"https://pith.science/pith/GP3UHDERY3BNWBDAWH3R2NDNHM/action/replication_record"}},"created_at":"2026-07-05T08:43:35.255004+00:00","updated_at":"2026-07-05T08:43:35.255004+00:00"}