{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:33N5IHA3VKDU3FL6EUNCFYNOAY","short_pith_number":"pith:33N5IHA3","schema_version":"1.0","canonical_sha256":"dedbd41c1baa874d957e251a22e1ae06328bba26af5b0b277f33bd5e8c2d23b0","source":{"kind":"arxiv","id":"2402.07872","version":1},"attestation_state":"computed","paper":{"title":"PIVOT: Iterative Visual Prompting Elicits Actionable Knowledge for VLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Andy Zeng, Annie Xie, Ayzaan Wahid, Brian Ichter, Chelsea Finn, Danny Driess, Fei Xia, Ishita Dasgupta, Jacky Liang, Karol Hausman, Kuang-Huei Lee, Nicolas Heess, Peng Xu, Quan Vuong, Sean Kirmani, Sergey Levine, Soroush Nasiriany, Ted Xiao, Tingnan Zhang, Tsang-Wei Edward Lee, Wenhao Yu, Yuke Zhu, Zhuo Xu","submitted_at":"2024-02-12T18:33:47Z","abstract_excerpt":"Vision language models (VLMs) have shown impressive capabilities across a variety of tasks, from logical reasoning to visual understanding. This opens the door to richer interaction with the world, for example robotic control. However, VLMs produce only textual outputs, while robotic control and other spatial tasks require outputting continuous coordinates, actions, or trajectories. How can we enable VLMs to handle such settings without fine-tuning on task-specific data?\n  In this paper, we propose a novel visual prompting approach for VLMs that we call Prompting with Iterative Visual Optimiza"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.07872","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-02-12T18:33:47Z","cross_cats_sorted":["cs.CL","cs.CV","cs.LG"],"title_canon_sha256":"f5ebb3fb981af6e263d5978ca6632afbe6b134b2768d18d64eab8895c8d55eb0","abstract_canon_sha256":"d7196323cbb364cef9a5b7ede7989f44330d78bbe737f85b32a73df165396ffc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:44:13.655537Z","signature_b64":"TA1LooZL9bKNp/ZMCz2GSx5AAQsRioiC+n1ngSJ+MZnP8vg6qQeAbcrqxofZL/HBON3tM348nwMEWFqjw5eVAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dedbd41c1baa874d957e251a22e1ae06328bba26af5b0b277f33bd5e8c2d23b0","last_reissued_at":"2026-07-05T07:44:13.655029Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:44:13.655029Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PIVOT: Iterative Visual Prompting Elicits Actionable Knowledge for VLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Andy Zeng, Annie Xie, Ayzaan Wahid, Brian Ichter, Chelsea Finn, Danny Driess, Fei Xia, Ishita Dasgupta, Jacky Liang, Karol Hausman, Kuang-Huei Lee, Nicolas Heess, Peng Xu, Quan Vuong, Sean Kirmani, Sergey Levine, Soroush Nasiriany, Ted Xiao, Tingnan Zhang, Tsang-Wei Edward Lee, Wenhao Yu, Yuke Zhu, Zhuo Xu","submitted_at":"2024-02-12T18:33:47Z","abstract_excerpt":"Vision language models (VLMs) have shown impressive capabilities across a variety of tasks, from logical reasoning to visual understanding. This opens the door to richer interaction with the world, for example robotic control. However, VLMs produce only textual outputs, while robotic control and other spatial tasks require outputting continuous coordinates, actions, or trajectories. How can we enable VLMs to handle such settings without fine-tuning on task-specific data?\n  In this paper, we propose a novel visual prompting approach for VLMs that we call Prompting with Iterative Visual Optimiza"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.07872","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.07872/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.07872","created_at":"2026-07-05T07:44:13.655089+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.07872v1","created_at":"2026-07-05T07:44:13.655089+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.07872","created_at":"2026-07-05T07:44:13.655089+00:00"},{"alias_kind":"pith_short_12","alias_value":"33N5IHA3VKDU","created_at":"2026-07-05T07:44:13.655089+00:00"},{"alias_kind":"pith_short_16","alias_value":"33N5IHA3VKDU3FL6","created_at":"2026-07-05T07:44:13.655089+00:00"},{"alias_kind":"pith_short_8","alias_value":"33N5IHA3","created_at":"2026-07-05T07:44:13.655089+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25880","citing_title":"USS: Unified Spatial-Semantic Prompts for Embodied Visual Tracking with Latent Dynamics Learning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26423","citing_title":"CoStream: Composing Simple Behaviors for Generalizable Complex Manipulation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19340","citing_title":"ZeroDex: Zero-Shot Long-Horizon Dexterous Manipulation via Multi-View 3D-Grounded VLM Reasoning","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13675","citing_title":"Improving Robotic Generalist Policies via Flow Reversal Steering","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31329","citing_title":"3D HAMSTER: Bridging Planning and Control in Hierarchical Vision Language Action Models through 3D Trajectory Guidance","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26423","citing_title":"CoStream: Composing Simple Behaviors for Generalizable Complex Manipulation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31329","citing_title":"3D HAMSTER: Bridging Planning and Control in Hierarchical Vision Language Action Models through 3D Trajectory Guidance","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29089","citing_title":"TAP-VLA: Tactile Annotation Prompting for Vision Language Action Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2503.07557","citing_title":"AutoSpatial: Visual-Language Reasoning for Social Robot Navigation through Efficient Spatial Reasoning Learning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2504.16054","citing_title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2403.01823","citing_title":"RT-H: Action Hierarchies Using Language","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2409.01652","citing_title":"ReKep: Spatio-Temporal Reasoning of Relational Keypoint Constraints for Robotic Manipulation","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2502.19417","citing_title":"Hi Robot: Open-Ended Instruction Following with Hierarchical Vision-Language-Action Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2412.10345","citing_title":"TraceVLA: Visual Trace Prompting Enhances Spatial-Temporal Awareness for Generalist Robotic Policies","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08509","citing_title":"Visually-grounded Humanoid Agents","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05498","citing_title":"JailWAM: Jailbreaking World Action Models in Robot Control","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/33N5IHA3VKDU3FL6EUNCFYNOAY","json":"https://pith.science/pith/33N5IHA3VKDU3FL6EUNCFYNOAY.json","graph_json":"https://pith.science/api/pith-number/33N5IHA3VKDU3FL6EUNCFYNOAY/graph.json","events_json":"https://pith.science/api/pith-number/33N5IHA3VKDU3FL6EUNCFYNOAY/events.json","paper":"https://pith.science/paper/33N5IHA3"},"agent_actions":{"view_html":"https://pith.science/pith/33N5IHA3VKDU3FL6EUNCFYNOAY","download_json":"https://pith.science/pith/33N5IHA3VKDU3FL6EUNCFYNOAY.json","view_paper":"https://pith.science/paper/33N5IHA3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.07872&json=true","fetch_graph":"https://pith.science/api/pith-number/33N5IHA3VKDU3FL6EUNCFYNOAY/graph.json","fetch_events":"https://pith.science/api/pith-number/33N5IHA3VKDU3FL6EUNCFYNOAY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/33N5IHA3VKDU3FL6EUNCFYNOAY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/33N5IHA3VKDU3FL6EUNCFYNOAY/action/storage_attestation","attest_author":"https://pith.science/pith/33N5IHA3VKDU3FL6EUNCFYNOAY/action/author_attestation","sign_citation":"https://pith.science/pith/33N5IHA3VKDU3FL6EUNCFYNOAY/action/citation_signature","submit_replication":"https://pith.science/pith/33N5IHA3VKDU3FL6EUNCFYNOAY/action/replication_record"}},"created_at":"2026-07-05T07:44:13.655089+00:00","updated_at":"2026-07-05T07:44:13.655089+00:00"}