{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NAQCE2ZP2BJ6BBI3DKMKS4IM35","short_pith_number":"pith:NAQCE2ZP","schema_version":"1.0","canonical_sha256":"6820226b2fd053e0851b1a98a9710cdf470ed7cd2ad6c4d6981de2f0cda880c5","source":{"kind":"arxiv","id":"2406.10721","version":1},"attestation_state":"computed","paper":{"title":"RoboPoint: A Vision-Language Model for Spatial Affordance Prediction for Robotics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.RO","authors_text":"Adithyavairavan Murali, Arsalan Mousavian, Dieter Fox, Jiafei Duan, Ranjay Krishna, Valts Blukis, Wentao Yuan, Wilbert Pumacay","submitted_at":"2024-06-15T19:22:51Z","abstract_excerpt":"From rearranging objects on a table to putting groceries into shelves, robots must plan precise action points to perform tasks accurately and reliably. In spite of the recent adoption of vision language models (VLMs) to control robot behavior, VLMs struggle to precisely articulate robot actions using language. We introduce an automatic synthetic data generation pipeline that instruction-tunes VLMs to robotic domains and needs. Using the pipeline, we train RoboPoint, a VLM that predicts image keypoint affordances given language instructions. Compared to alternative approaches, our method requir"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.10721","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-06-15T19:22:51Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"fc86982d82619e8306fcf6576d180ce6b59f647de2944285d3ea85858a474f11","abstract_canon_sha256":"c00ba65e93cf02a6c2a2aec126a54e7235219e9626e5da4d6d5c8c2752456fde"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:32:20.907744Z","signature_b64":"1UioDiqF6uQRdEVPOgUEuR2yaFX5i/Xa7c8h8eKa8Z4qWWG6DM5kd0oylWFt4butsHyO1XRthGqdIjtwRalUBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6820226b2fd053e0851b1a98a9710cdf470ed7cd2ad6c4d6981de2f0cda880c5","last_reissued_at":"2026-07-05T08:32:20.907240Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:32:20.907240Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RoboPoint: A Vision-Language Model for Spatial Affordance Prediction for Robotics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.RO","authors_text":"Adithyavairavan Murali, Arsalan Mousavian, Dieter Fox, Jiafei Duan, Ranjay Krishna, Valts Blukis, Wentao Yuan, Wilbert Pumacay","submitted_at":"2024-06-15T19:22:51Z","abstract_excerpt":"From rearranging objects on a table to putting groceries into shelves, robots must plan precise action points to perform tasks accurately and reliably. In spite of the recent adoption of vision language models (VLMs) to control robot behavior, VLMs struggle to precisely articulate robot actions using language. We introduce an automatic synthetic data generation pipeline that instruction-tunes VLMs to robotic domains and needs. Using the pipeline, we train RoboPoint, a VLM that predicts image keypoint affordances given language instructions. Compared to alternative approaches, our method requir"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.10721","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.10721/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.10721","created_at":"2026-07-05T08:32:20.907304+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.10721v1","created_at":"2026-07-05T08:32:20.907304+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.10721","created_at":"2026-07-05T08:32:20.907304+00:00"},{"alias_kind":"pith_short_12","alias_value":"NAQCE2ZP2BJ6","created_at":"2026-07-05T08:32:20.907304+00:00"},{"alias_kind":"pith_short_16","alias_value":"NAQCE2ZP2BJ6BBI3","created_at":"2026-07-05T08:32:20.907304+00:00"},{"alias_kind":"pith_short_8","alias_value":"NAQCE2ZP","created_at":"2026-07-05T08:32:20.907304+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":39,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26423","citing_title":"CoStream: Composing Simple Behaviors for Generalizable Complex Manipulation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22476","citing_title":"CVSBench: A Comprehensive Benchmark for Cross-view Spatial Reasoning and Dreaming","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20905","citing_title":"Vesta: A Generalist Embodied Reasoning Model","ref_index":148,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19340","citing_title":"ZeroDex: Zero-Shot Long-Horizon Dexterous Manipulation via Multi-View 3D-Grounded VLM Reasoning","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13515","citing_title":"MaskWAM: Unifying Mask Prompting and Prediction for World-Action Models","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11770","citing_title":"SVoT: State-aware Visualization-of-Thought for Spatial Reasoning via Reinforcement Learning","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08520","citing_title":"Two Bridges, One Pathway: From VLMs to Generalizable VLAs with Embodied Trajectory-Coupled Data","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00881","citing_title":"OmniView-Space: Reinforcing Spatial Reasoning via Multi-Perspective Spatial Mapping","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06761","citing_title":"AxisGuide: Grounding Robot Action Coordinate System in RGB Observations for Robust Visuomotor Manipulation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03240","citing_title":"GeoAlign: Beyond Semantics with State-Guided Spatial Alignment in VLA Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30877","citing_title":"Wall-OSS-0.5 Technical Report","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26423","citing_title":"CoStream: Composing Simple Behaviors for Generalizable Complex Manipulation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14923","citing_title":"SceneParser: Hierarchical Scene Parsing for Visual Semantics Understanding","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24203","citing_title":"Afford-VLA: Action-Aligned Visual Planning via Internalized Affordance","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29089","citing_title":"TAP-VLA: Tactile Annotation Prompting for Vision Language Action Models","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.15753","citing_title":"RoboPIN: Grounded Embodied Reasoning via Pinned Chain-of-Thought","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29267","citing_title":"Enhancing Part-Level Point Grounding for Any Open-Source MLLMs","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25813","citing_title":"Extending Embodied Question Answering from Perception to Decision","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28548","citing_title":"GEM: Generative Supervision Helps Embodied Intelligence","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01247","citing_title":"Where to Look: Can Foundation Models Reach a Target Viewpoint Through Active Exploration?","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11324","citing_title":"Embodied-R1.5: Evolving Physical Intelligence via Embodied Foundation Models","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02239","citing_title":"LACY: A Vision-Language Model-based Language-Action Cycle for Self-Improving Robotic Manipulation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2405.14093","citing_title":"A Survey on Vision-Language-Action Models for Embodied AI","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21133","citing_title":"Humanoid Whole-Body Manipulation via Active Spatial Brain and Generalizable Action Cerebellum","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2507.16815","citing_title":"ThinkAct: Vision-Language-Action Reasoning via Reinforced Visual Latent Planning","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NAQCE2ZP2BJ6BBI3DKMKS4IM35","json":"https://pith.science/pith/NAQCE2ZP2BJ6BBI3DKMKS4IM35.json","graph_json":"https://pith.science/api/pith-number/NAQCE2ZP2BJ6BBI3DKMKS4IM35/graph.json","events_json":"https://pith.science/api/pith-number/NAQCE2ZP2BJ6BBI3DKMKS4IM35/events.json","paper":"https://pith.science/paper/NAQCE2ZP"},"agent_actions":{"view_html":"https://pith.science/pith/NAQCE2ZP2BJ6BBI3DKMKS4IM35","download_json":"https://pith.science/pith/NAQCE2ZP2BJ6BBI3DKMKS4IM35.json","view_paper":"https://pith.science/paper/NAQCE2ZP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.10721&json=true","fetch_graph":"https://pith.science/api/pith-number/NAQCE2ZP2BJ6BBI3DKMKS4IM35/graph.json","fetch_events":"https://pith.science/api/pith-number/NAQCE2ZP2BJ6BBI3DKMKS4IM35/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NAQCE2ZP2BJ6BBI3DKMKS4IM35/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NAQCE2ZP2BJ6BBI3DKMKS4IM35/action/storage_attestation","attest_author":"https://pith.science/pith/NAQCE2ZP2BJ6BBI3DKMKS4IM35/action/author_attestation","sign_citation":"https://pith.science/pith/NAQCE2ZP2BJ6BBI3DKMKS4IM35/action/citation_signature","submit_replication":"https://pith.science/pith/NAQCE2ZP2BJ6BBI3DKMKS4IM35/action/replication_record"}},"created_at":"2026-07-05T08:32:20.907304+00:00","updated_at":"2026-07-05T08:32:20.907304+00:00"}