{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HMBANQW4IAFXOGN3CDCLYWR4Y2","short_pith_number":"pith:HMBANQW4","schema_version":"1.0","canonical_sha256":"3b0206c2dc400b7719bb10c4bc5a3cc6a88a441784c640e5435543f83530cbbf","source":{"kind":"arxiv","id":"2311.17842","version":2},"attestation_state":"computed","paper":{"title":"Look Before You Leap: Unveiling the Power of GPT-4V in Robotic Vision-Language Planning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Fanqi Lin, Li Yi, Tong Zhang, Yang Gao, Yingdong Hu","submitted_at":"2023-11-29T17:46:25Z","abstract_excerpt":"In this study, we are interested in imbuing robots with the capability of physically-grounded task planning. Recent advancements have shown that large language models (LLMs) possess extensive knowledge useful in robotic tasks, especially in reasoning and planning. However, LLMs are constrained by their lack of world grounding and dependence on external affordance models to perceive environmental information, which cannot jointly reason with LLMs. We argue that a task planner should be an inherently grounded, unified multimodal system. To this end, we introduce Robotic Vision-Language Planning "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.17842","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2023-11-29T17:46:25Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV","cs.LG"],"title_canon_sha256":"081e01379bf5f355b209fc82202f15eb547ce11527d20224a592cc462652e93a","abstract_canon_sha256":"7d91557095a8b196fc307b5c569b5306564aa5741f29d6734f3fe45f72189784"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:27:53.204152Z","signature_b64":"LKy+3m7JnBoZ+NSha850tgGZf9fZgvWugAdUf2DOGEotQwkfpajWm7luhSeakEpXgbRHNHfxABfO2XXeANRgBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3b0206c2dc400b7719bb10c4bc5a3cc6a88a441784c640e5435543f83530cbbf","last_reissued_at":"2026-07-05T07:27:53.203603Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:27:53.203603Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Look Before You Leap: Unveiling the Power of GPT-4V in Robotic Vision-Language Planning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Fanqi Lin, Li Yi, Tong Zhang, Yang Gao, Yingdong Hu","submitted_at":"2023-11-29T17:46:25Z","abstract_excerpt":"In this study, we are interested in imbuing robots with the capability of physically-grounded task planning. Recent advancements have shown that large language models (LLMs) possess extensive knowledge useful in robotic tasks, especially in reasoning and planning. However, LLMs are constrained by their lack of world grounding and dependence on external affordance models to perceive environmental information, which cannot jointly reason with LLMs. We argue that a task planner should be an inherently grounded, unified multimodal system. To this end, we introduce Robotic Vision-Language Planning "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.17842","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.17842/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.17842","created_at":"2026-07-05T07:27:53.203665+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.17842v2","created_at":"2026-07-05T07:27:53.203665+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.17842","created_at":"2026-07-05T07:27:53.203665+00:00"},{"alias_kind":"pith_short_12","alias_value":"HMBANQW4IAFX","created_at":"2026-07-05T07:27:53.203665+00:00"},{"alias_kind":"pith_short_16","alias_value":"HMBANQW4IAFXOGN3","created_at":"2026-07-05T07:27:53.203665+00:00"},{"alias_kind":"pith_short_8","alias_value":"HMBANQW4","created_at":"2026-07-05T07:27:53.203665+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06706","citing_title":"Vision Language Action (VLA) Models for Unmanned Aerial Robotics and Bimanual Manipulation: A Review","ref_index":113,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26423","citing_title":"CoStream: Composing Simple Behaviors for Generalizable Complex Manipulation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22488","citing_title":"SCOPE: Evolving Symbolic World for Planning in Open-Ended Environments","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11770","citing_title":"SVoT: State-aware Visualization-of-Thought for Spatial Reasoning via Reinforcement Learning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00985","citing_title":"Make Your VLA More Robust Without More Data By Interleaving Motion Planning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26423","citing_title":"CoStream: Composing Simple Behaviors for Generalizable Complex Manipulation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2310.02635","citing_title":"Reinforcement Learning with Foundation Priors: Let the Embodied Agent Efficiently Learn on Its Own","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10446","citing_title":"VeriGraph: Scene Graphs for Execution Verifiable Robot Planning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2504.16054","citing_title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15975","citing_title":"Learning Bilevel Policies over Symbolic World Models for Long-Horizon Planning","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2506.06856","citing_title":"Vision-EKIPL: External Knowledge-Infused Policy Learning for Visual Reasoning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2402.15852","citing_title":"NaVid: Video-based VLM Plans the Next Step for Vision-and-Language Navigation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2508.13073","citing_title":"Large VLM-based Vision-Language-Action Models for Robotic Manipulation: A Survey","ref_index":154,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18203","citing_title":"SkillWrapper: Generative Predicate Invention for Task-level Robot Planning","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2409.01652","citing_title":"ReKep: Spatio-Temporal Reasoning of Relational Keypoint Constraints for Robotic Manipulation","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2502.19417","citing_title":"Hi Robot: Open-Ended Instruction Following with Hierarchical Vision-Language-Action Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2603.25044","citing_title":"ThermoAct:Thermal-Aware Vision-Language-Action Models for Robotic Perception and Decision-Making","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05855","citing_title":"DexVLA: Vision-Language Model with Plug-In Diffusion Expert for General Robot Control","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25788","citing_title":"KinDER: A Physical Reasoning Benchmark for Robot Learning and Planning","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22152","citing_title":"dWorldEval: Scalable Robotic Policy Evaluation via Discrete Diffusion World Model","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07774","citing_title":"RoboAgent: Chaining Basic Capabilities for Embodied Task Planning","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HMBANQW4IAFXOGN3CDCLYWR4Y2","json":"https://pith.science/pith/HMBANQW4IAFXOGN3CDCLYWR4Y2.json","graph_json":"https://pith.science/api/pith-number/HMBANQW4IAFXOGN3CDCLYWR4Y2/graph.json","events_json":"https://pith.science/api/pith-number/HMBANQW4IAFXOGN3CDCLYWR4Y2/events.json","paper":"https://pith.science/paper/HMBANQW4"},"agent_actions":{"view_html":"https://pith.science/pith/HMBANQW4IAFXOGN3CDCLYWR4Y2","download_json":"https://pith.science/pith/HMBANQW4IAFXOGN3CDCLYWR4Y2.json","view_paper":"https://pith.science/paper/HMBANQW4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.17842&json=true","fetch_graph":"https://pith.science/api/pith-number/HMBANQW4IAFXOGN3CDCLYWR4Y2/graph.json","fetch_events":"https://pith.science/api/pith-number/HMBANQW4IAFXOGN3CDCLYWR4Y2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HMBANQW4IAFXOGN3CDCLYWR4Y2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HMBANQW4IAFXOGN3CDCLYWR4Y2/action/storage_attestation","attest_author":"https://pith.science/pith/HMBANQW4IAFXOGN3CDCLYWR4Y2/action/author_attestation","sign_citation":"https://pith.science/pith/HMBANQW4IAFXOGN3CDCLYWR4Y2/action/citation_signature","submit_replication":"https://pith.science/pith/HMBANQW4IAFXOGN3CDCLYWR4Y2/action/replication_record"}},"created_at":"2026-07-05T07:27:53.203665+00:00","updated_at":"2026-07-05T07:27:53.203665+00:00"}