{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NNKI34S7SAK4R6JCDXLJPUEGAQ","short_pith_number":"pith:NNKI34S7","schema_version":"1.0","canonical_sha256":"6b548df25f9015c8f9221dd697d086040d511a872677bae3aab5d7d60ec28c23","source":{"kind":"arxiv","id":"2508.09071","version":2},"attestation_state":"computed","paper":{"title":"GeoVLA: Empowering 3D Representations in Vision-Language-Action Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Bin Xie, Hao Shi, Jiale Cao, Lin Sun, Tiancai Wang, Yingfei Liu","submitted_at":"2025-08-12T16:46:05Z","abstract_excerpt":"Vision-Language-Action (VLA) models have emerged as a promising approach for enabling robots to follow language instructions and predict corresponding actions. However, current VLA models mainly rely on 2D visual inputs, neglecting the rich geometric information in the 3D physical world, which limits their spatial awareness and adaptability. In this paper, we present GeoVLA, a novel VLA framework that effectively integrates 3D information to advance robotic manipulation. It uses a vision-language model (VLM) to process images and language instructions,extracting fused vision-language embedding"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.09071","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.RO","submitted_at":"2025-08-12T16:46:05Z","cross_cats_sorted":[],"title_canon_sha256":"ed0c2e59b341d7c1376d3b6ee246fd57f70059abb62e7b39b1a20dfa4b299f60","abstract_canon_sha256":"c5b045bcfc9446b722fa7bf685d212b64776619496da368609b70c548f70189a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:53:08.566125Z","signature_b64":"a7QbT5Dm7DOzB+loq0bBaXEmsdWYYWUZBp5YzcnSY7S6w+RTbfcCBSsaaH8+/8QuGeeWvyRKsVG30FvuCaLOBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6b548df25f9015c8f9221dd697d086040d511a872677bae3aab5d7d60ec28c23","last_reissued_at":"2026-07-05T11:53:08.565667Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:53:08.565667Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GeoVLA: Empowering 3D Representations in Vision-Language-Action Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Bin Xie, Hao Shi, Jiale Cao, Lin Sun, Tiancai Wang, Yingfei Liu","submitted_at":"2025-08-12T16:46:05Z","abstract_excerpt":"Vision-Language-Action (VLA) models have emerged as a promising approach for enabling robots to follow language instructions and predict corresponding actions. However, current VLA models mainly rely on 2D visual inputs, neglecting the rich geometric information in the 3D physical world, which limits their spatial awareness and adaptability. In this paper, we present GeoVLA, a novel VLA framework that effectively integrates 3D information to advance robotic manipulation. It uses a vision-language model (VLM) to process images and language instructions,extracting fused vision-language embedding"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.09071","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.09071/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.09071","created_at":"2026-07-05T11:53:08.565715+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.09071v2","created_at":"2026-07-05T11:53:08.565715+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.09071","created_at":"2026-07-05T11:53:08.565715+00:00"},{"alias_kind":"pith_short_12","alias_value":"NNKI34S7SAK4","created_at":"2026-07-05T11:53:08.565715+00:00"},{"alias_kind":"pith_short_16","alias_value":"NNKI34S7SAK4R6JC","created_at":"2026-07-05T11:53:08.565715+00:00"},{"alias_kind":"pith_short_8","alias_value":"NNKI34S7","created_at":"2026-07-05T11:53:08.565715+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":27,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17046","citing_title":"Geometric Action Model for Robot Policy Learning","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13394","citing_title":"GeoHAT: Geometry-Adaptive Hybrid Action Transformer for Mobile Manipulation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10568","citing_title":"VeriSpace: Spatially Grounded Action Verification for Vision-Language-Action Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09827","citing_title":"MemoryVLA++: Temporal Modeling via Memory and Imagination in Vision-Language-Action Models","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08288","citing_title":"MotionVLA: Injecting Geometric Motion into Vision-Language-Action Model","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06556","citing_title":"Robots Need More than VLA and World Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04436","citing_title":"3DThinkVLA: Endowing Vision-Language-Action Models with Latent 3D Priors via 3D-Thinking-Guided Co-training","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03240","citing_title":"GeoAlign: Beyond Semantics with State-Guided Spatial Alignment in VLA Models","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02274","citing_title":"Dexterity-BEV: Aligning 3D World and Actions for Generalizable Robot Policies Learning","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12369","citing_title":"GuidedVLA: Specifying Task-Relevant Factors via Plug-and-Play Action Attention Specialization","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30484","citing_title":"ELAN4D: Embodiment-Centric 4D Supervision for Vision-Language-Action Models via Plug-and-Play Adaptation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21414","citing_title":"PointACT: Vision-Language-Action Models with Multi-Scale Point-Action Interaction","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2510.03827","citing_title":"LIBERO-PRO: Towards Robust and Fair Evaluation of Vision-Language-Action Models Beyond Memorization","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2601.18692","citing_title":"A Pragmatic VLA Foundation Model","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2508.19236","citing_title":"MemoryVLA: Perceptual-Cognitive Memory in Vision-Language-Action Models for Robotic Manipulation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11832","citing_title":"Learning Action Manifold with Multi-view Latent Priors for Robotic Manipulation","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12162","citing_title":"X-Imitator: Spatial-Aware Imitation Learning via Bidirectional Action-Pose Interaction","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12369","citing_title":"GuidedVLA: Specifying Task-Relevant Factors via Plug-and-Play Action Attention Specialization","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26694","citing_title":"Unified 4D World Action Modeling from Video Priors with Asynchronous Denoising","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10485","citing_title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26694","citing_title":"Unified 4D World Action Modeling from Video Priors with Asynchronous Denoising","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05126","citing_title":"ConsisVLA-4D: Advancing Spatiotemporal Consistency in Efficient 3D-Perception and 4D-Reasoning for Robotic Manipulation","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12908","citing_title":"Robotic Manipulation is Vision-to-Geometry Mapping ($f(v) \\rightarrow G$): Vision-Geometry Backbones over Language and Video Models","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05484","citing_title":"CoEnv: Driving Embodied Multi-Agent Collaboration via Compositional Environment","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04834","citing_title":"E-VLA: Event-Augmented Vision-Language-Action Model for Dark and Blurred Scenes","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NNKI34S7SAK4R6JCDXLJPUEGAQ","json":"https://pith.science/pith/NNKI34S7SAK4R6JCDXLJPUEGAQ.json","graph_json":"https://pith.science/api/pith-number/NNKI34S7SAK4R6JCDXLJPUEGAQ/graph.json","events_json":"https://pith.science/api/pith-number/NNKI34S7SAK4R6JCDXLJPUEGAQ/events.json","paper":"https://pith.science/paper/NNKI34S7"},"agent_actions":{"view_html":"https://pith.science/pith/NNKI34S7SAK4R6JCDXLJPUEGAQ","download_json":"https://pith.science/pith/NNKI34S7SAK4R6JCDXLJPUEGAQ.json","view_paper":"https://pith.science/paper/NNKI34S7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.09071&json=true","fetch_graph":"https://pith.science/api/pith-number/NNKI34S7SAK4R6JCDXLJPUEGAQ/graph.json","fetch_events":"https://pith.science/api/pith-number/NNKI34S7SAK4R6JCDXLJPUEGAQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NNKI34S7SAK4R6JCDXLJPUEGAQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NNKI34S7SAK4R6JCDXLJPUEGAQ/action/storage_attestation","attest_author":"https://pith.science/pith/NNKI34S7SAK4R6JCDXLJPUEGAQ/action/author_attestation","sign_citation":"https://pith.science/pith/NNKI34S7SAK4R6JCDXLJPUEGAQ/action/citation_signature","submit_replication":"https://pith.science/pith/NNKI34S7SAK4R6JCDXLJPUEGAQ/action/replication_record"}},"created_at":"2026-07-05T11:53:08.565715+00:00","updated_at":"2026-07-05T11:53:08.565715+00:00"}