{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:E6MOC4LLOZXY3BS25PCA2BJRNH","short_pith_number":"pith:E6MOC4LL","schema_version":"1.0","canonical_sha256":"2798e1716b766f8d865aebc40d053169d1cdec4ab7862ee690c1e098260f3fff","source":{"kind":"arxiv","id":"2406.01584","version":3},"attestation_state":"computed","paper":{"title":"SpatialRGPT: Grounded Spatial Reasoning in Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"An-Chieh Cheng, Hongxu Yin, Jan Kautz, Qiushan Guo, Ruihan Yang, Sifei Liu, Xiaolong Wang, Yang Fu","submitted_at":"2024-06-03T17:59:06Z","abstract_excerpt":"Vision Language Models (VLMs) have demonstrated remarkable performance in 2D vision and language tasks. However, their ability to reason about spatial arrangements remains limited. In this work, we introduce Spatial Region GPT (SpatialRGPT) to enhance VLMs' spatial perception and reasoning capabilities. SpatialRGPT advances VLMs' spatial understanding through two key innovations: (1) a data curation pipeline that enables effective learning of regional representation from 3D scene graphs, and (2) a flexible plugin module for integrating depth information into the visual encoder of existing VLMs"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.01584","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-03T17:59:06Z","cross_cats_sorted":[],"title_canon_sha256":"33546d9eae62058abbe9d5f8833a42d30991d14eab354c4bb5c957c902a3d5ab","abstract_canon_sha256":"55699ee1793b081fde323df8d0eaeb360081aa4051438cfa65eb3cb6ba14c6ab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:23.641253Z","signature_b64":"q4CXTQZegu/YHHmnCBr9U6B4QMGrAg51ud2nB3em6tJIAcmu2lcRyNuZ0Qwhj3IsxBGuHgwXKLpGiMEyH6FtCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2798e1716b766f8d865aebc40d053169d1cdec4ab7862ee690c1e098260f3fff","last_reissued_at":"2026-07-05T09:20:23.640756Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:23.640756Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SpatialRGPT: Grounded Spatial Reasoning in Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"An-Chieh Cheng, Hongxu Yin, Jan Kautz, Qiushan Guo, Ruihan Yang, Sifei Liu, Xiaolong Wang, Yang Fu","submitted_at":"2024-06-03T17:59:06Z","abstract_excerpt":"Vision Language Models (VLMs) have demonstrated remarkable performance in 2D vision and language tasks. However, their ability to reason about spatial arrangements remains limited. In this work, we introduce Spatial Region GPT (SpatialRGPT) to enhance VLMs' spatial perception and reasoning capabilities. SpatialRGPT advances VLMs' spatial understanding through two key innovations: (1) a data curation pipeline that enables effective learning of regional representation from 3D scene graphs, and (2) a flexible plugin module for integrating depth information into the visual encoder of existing VLMs"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.01584","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.01584/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.01584","created_at":"2026-07-05T09:20:23.640815+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.01584v3","created_at":"2026-07-05T09:20:23.640815+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.01584","created_at":"2026-07-05T09:20:23.640815+00:00"},{"alias_kind":"pith_short_12","alias_value":"E6MOC4LLOZXY","created_at":"2026-07-05T09:20:23.640815+00:00"},{"alias_kind":"pith_short_16","alias_value":"E6MOC4LLOZXY3BS2","created_at":"2026-07-05T09:20:23.640815+00:00"},{"alias_kind":"pith_short_8","alias_value":"E6MOC4LL","created_at":"2026-07-05T09:20:23.640815+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17539","citing_title":"Reinforcing Dual-Path Reasoning in Spatial Vision Language Models","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07861","citing_title":"The Last Visible Pixel: Probing Fine-Scale Perception in Vision-Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05677","citing_title":"LongSpace: Exploring Long-Horizon Spatial Memory from Perception to Recall in Video","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02842","citing_title":"Spectral-Progressive Thought Flow for Lightweight Multimodal Reasoning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18746","citing_title":"ESI-Bench: Towards Embodied Spatial Intelligence that Closes the Perception-Action Loop","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20448","citing_title":"Do Vision-Language Models Understand 3D Scenes or Just Catalogue Objects?","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30557","citing_title":"Seeing Isn't Knowing: Do VLMs Know When Not to Answer Spatial Questions (and Why)?","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22219","citing_title":"Lost in Aggregation: A Multi-Scale Diagnostic Benchmark for LLM Spatial Navigation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20448","citing_title":"Do Vision-Language Models Understand 3D Scenes or Just Catalogue Objects?","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18746","citing_title":"ESI-Bench: Towards Embodied Spatial Intelligence that Closes the Perception-Action Loop","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18621","citing_title":"CrossView Suite: Harnessing Cross-view Spatial Intelligence of MLLMs with Dataset, Model and Benchmark","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2508.13998","citing_title":"Embodied-R1: Reinforced Embodied Reasoning for General Robotic Manipulation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02627","citing_title":"DecompSR: A dataset for decomposed analyses of compositional multihop spatial reasoning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2501.07542","citing_title":"Imagine while Reasoning in Space: Multimodal Visualization-of-Thought","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2601.22228","citing_title":"Lost in Space? Vision-Language Models Struggle with Relative Camera Pose Estimation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08096","citing_title":"TrianguLang: Geometry-Aware Semantic Consensus for Pose-Free 3D Localization","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03660","citing_title":"TableVision: A Large-Scale Benchmark for Spatially Grounded Reasoning over Complex Hierarchical Tables","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07592","citing_title":"Spatio-Temporal Grounding of Large Language Models from Perception Streams","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E6MOC4LLOZXY3BS25PCA2BJRNH","json":"https://pith.science/pith/E6MOC4LLOZXY3BS25PCA2BJRNH.json","graph_json":"https://pith.science/api/pith-number/E6MOC4LLOZXY3BS25PCA2BJRNH/graph.json","events_json":"https://pith.science/api/pith-number/E6MOC4LLOZXY3BS25PCA2BJRNH/events.json","paper":"https://pith.science/paper/E6MOC4LL"},"agent_actions":{"view_html":"https://pith.science/pith/E6MOC4LLOZXY3BS25PCA2BJRNH","download_json":"https://pith.science/pith/E6MOC4LLOZXY3BS25PCA2BJRNH.json","view_paper":"https://pith.science/paper/E6MOC4LL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.01584&json=true","fetch_graph":"https://pith.science/api/pith-number/E6MOC4LLOZXY3BS25PCA2BJRNH/graph.json","fetch_events":"https://pith.science/api/pith-number/E6MOC4LLOZXY3BS25PCA2BJRNH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E6MOC4LLOZXY3BS25PCA2BJRNH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E6MOC4LLOZXY3BS25PCA2BJRNH/action/storage_attestation","attest_author":"https://pith.science/pith/E6MOC4LLOZXY3BS25PCA2BJRNH/action/author_attestation","sign_citation":"https://pith.science/pith/E6MOC4LLOZXY3BS25PCA2BJRNH/action/citation_signature","submit_replication":"https://pith.science/pith/E6MOC4LLOZXY3BS25PCA2BJRNH/action/replication_record"}},"created_at":"2026-07-05T09:20:23.640815+00:00","updated_at":"2026-07-05T09:20:23.640815+00:00"}