{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MV46K5N6UM3KJKIR3BOSWSPKJJ","short_pith_number":"pith:MV46K5N6","schema_version":"1.0","canonical_sha256":"6579e575bea336a4a911d85d2b49ea4a7a60f8bc91e4c1da4660cd699ce141dc","source":{"kind":"arxiv","id":"2401.12168","version":1},"attestation_state":"computed","paper":{"title":"SpatialVLM: Endowing Vision-Language Models with Spatial Reasoning Capabilities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Boyuan Chen, Brian Ichter, Danny Driess, Dorsa Sadigh, Fei Xia, Leonidas Guibas, Pete Florence, Sean Kirmani, Zhuo Xu","submitted_at":"2024-01-22T18:01:01Z","abstract_excerpt":"Understanding and reasoning about spatial relationships is a fundamental capability for Visual Question Answering (VQA) and robotics. While Vision Language Models (VLM) have demonstrated remarkable performance in certain VQA benchmarks, they still lack capabilities in 3D spatial reasoning, such as recognizing quantitative relationships of physical objects like distances or size differences. We hypothesize that VLMs' limited spatial reasoning capability is due to the lack of 3D spatial knowledge in training data and aim to solve this problem by training VLMs with Internet-scale spatial reasonin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.12168","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-22T18:01:01Z","cross_cats_sorted":["cs.CL","cs.LG","cs.RO"],"title_canon_sha256":"68bb85a224eb63bc86b944b887e7e64730e9203fb1a4b1ff8f0b2db7b76e3db4","abstract_canon_sha256":"21ab30f07420e661392af14e6298174bb7400e5a2234a428b86ca00eeeea7101"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:36:16.363307Z","signature_b64":"URQOfD/Ag9HhAD1WVdY15u3SwKG0/I7n2dQ6MTcIyX6cp7dQnYSNcl5Vdid11iX7m+yscCxzfUyj7RLDve1bAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6579e575bea336a4a911d85d2b49ea4a7a60f8bc91e4c1da4660cd699ce141dc","last_reissued_at":"2026-07-05T07:36:16.362868Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:36:16.362868Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SpatialVLM: Endowing Vision-Language Models with Spatial Reasoning Capabilities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Boyuan Chen, Brian Ichter, Danny Driess, Dorsa Sadigh, Fei Xia, Leonidas Guibas, Pete Florence, Sean Kirmani, Zhuo Xu","submitted_at":"2024-01-22T18:01:01Z","abstract_excerpt":"Understanding and reasoning about spatial relationships is a fundamental capability for Visual Question Answering (VQA) and robotics. While Vision Language Models (VLM) have demonstrated remarkable performance in certain VQA benchmarks, they still lack capabilities in 3D spatial reasoning, such as recognizing quantitative relationships of physical objects like distances or size differences. We hypothesize that VLMs' limited spatial reasoning capability is due to the lack of 3D spatial knowledge in training data and aim to solve this problem by training VLMs with Internet-scale spatial reasonin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.12168","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.12168/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.12168","created_at":"2026-07-05T07:36:16.362922+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.12168v1","created_at":"2026-07-05T07:36:16.362922+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.12168","created_at":"2026-07-05T07:36:16.362922+00:00"},{"alias_kind":"pith_short_12","alias_value":"MV46K5N6UM3K","created_at":"2026-07-05T07:36:16.362922+00:00"},{"alias_kind":"pith_short_16","alias_value":"MV46K5N6UM3KJKIR","created_at":"2026-07-05T07:36:16.362922+00:00"},{"alias_kind":"pith_short_8","alias_value":"MV46K5N6","created_at":"2026-07-05T07:36:16.362922+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08724","citing_title":"Latent Memory Palace: Reasoning for Control as Autoregressive Variational Inference","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23565","citing_title":"HoloAgent-0: A Unified Embodied Agent Framework with 3D Spatial Memory","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17539","citing_title":"Reinforcing Dual-Path Reasoning in Spatial Vision Language Models","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05833","citing_title":"Learning Geometric Representations from Videos for Spatial Intelligent Multimodal Large Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05677","citing_title":"LongSpace: Exploring Long-Horizon Spatial Memory from Perception to Recall in Video","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03240","citing_title":"GeoAlign: Beyond Semantics with State-Guided Spatial Alignment in VLA Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03577","citing_title":"Eliciting Complex Spatial Reasoning in MLLMs through Wide-Baseline Matching","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00963","citing_title":"Reasmory: 3D Reconstruction as Explicit Memory for VLMs Spatial Reasoning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31200","citing_title":"Agentic RAG-VLM: Affordance-Aware Retrieval-Augmented Generation with Self-Reflective Planning for Robotic Grasping","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13169","citing_title":"PanoWorld: Towards Spatial Supersensing in 360$^\\circ$ Panorama World","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08096","citing_title":"TrianguLang: Geometry-Aware Semantic Consensus for Pose-Free 3D Localization","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13169","citing_title":"PanoWorld: Towards Spatial Supersensing in 360$^\\circ$ Panorama World","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00799","citing_title":"Multimodal Language Models Cannot Spot Spatial Inconsistencies","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25954","citing_title":"Fast Core Identification","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01694","citing_title":"Latent State Design for World Models under Sufficiency Constraints","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20570","citing_title":"Exploring Spatial Intelligence from a Generative Perspective","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MV46K5N6UM3KJKIR3BOSWSPKJJ","json":"https://pith.science/pith/MV46K5N6UM3KJKIR3BOSWSPKJJ.json","graph_json":"https://pith.science/api/pith-number/MV46K5N6UM3KJKIR3BOSWSPKJJ/graph.json","events_json":"https://pith.science/api/pith-number/MV46K5N6UM3KJKIR3BOSWSPKJJ/events.json","paper":"https://pith.science/paper/MV46K5N6"},"agent_actions":{"view_html":"https://pith.science/pith/MV46K5N6UM3KJKIR3BOSWSPKJJ","download_json":"https://pith.science/pith/MV46K5N6UM3KJKIR3BOSWSPKJJ.json","view_paper":"https://pith.science/paper/MV46K5N6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.12168&json=true","fetch_graph":"https://pith.science/api/pith-number/MV46K5N6UM3KJKIR3BOSWSPKJJ/graph.json","fetch_events":"https://pith.science/api/pith-number/MV46K5N6UM3KJKIR3BOSWSPKJJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MV46K5N6UM3KJKIR3BOSWSPKJJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MV46K5N6UM3KJKIR3BOSWSPKJJ/action/storage_attestation","attest_author":"https://pith.science/pith/MV46K5N6UM3KJKIR3BOSWSPKJJ/action/author_attestation","sign_citation":"https://pith.science/pith/MV46K5N6UM3KJKIR3BOSWSPKJJ/action/citation_signature","submit_replication":"https://pith.science/pith/MV46K5N6UM3KJKIR3BOSWSPKJJ/action/replication_record"}},"created_at":"2026-07-05T07:36:16.362922+00:00","updated_at":"2026-07-05T07:36:16.362922+00:00"}