{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3VDRO65GHSMFKOEJ4CG7NHO37B","short_pith_number":"pith:3VDRO65G","schema_version":"1.0","canonical_sha256":"dd47177ba63c98553889e08df69ddbf86e721acd4f2c0042145b08311d5d46a2","source":{"kind":"arxiv","id":"2501.01428","version":4},"attestation_state":"computed","paper":{"title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hengshuang Zhao, Jiaqi Wang, Ye Fang, Zhangyang Qi, Zhixiong Zhang","submitted_at":"2025-01-02T18:59:59Z","abstract_excerpt":"In recent years, 2D Vision-Language Models (VLMs) have made significant strides in image-text understanding tasks. However, their performance in 3D spatial comprehension, which is critical for embodied intelligence, remains limited. Recent advances have leveraged 3D point clouds and multi-view images as inputs, yielding promising results. However, we propose exploring a purely vision-based solution inspired by human perception, which merely relies on visual cues for 3D spatial understanding. This paper empirically investigates the limitations of VLMs in 3D spatial knowledge, revealing that the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.01428","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-01-02T18:59:59Z","cross_cats_sorted":[],"title_canon_sha256":"d6bc9f0f4663acb17b23ba591dda3e51e8fcc90118f6899f27d930093717594e","abstract_canon_sha256":"282786e288417fc5a4cbac735bcebe1074228664294d7b284505e2afda8dbc43"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:28:31.631343Z","signature_b64":"wp6qhL3isppSDL4Ysh+bTu+g2tYhbmypLJzIqbFhsV84HDHdeCWyYoujGqUrFLTK/qaVRy8h2I40r7YatuDZBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dd47177ba63c98553889e08df69ddbf86e721acd4f2c0042145b08311d5d46a2","last_reissued_at":"2026-07-05T10:28:31.630273Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:28:31.630273Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hengshuang Zhao, Jiaqi Wang, Ye Fang, Zhangyang Qi, Zhixiong Zhang","submitted_at":"2025-01-02T18:59:59Z","abstract_excerpt":"In recent years, 2D Vision-Language Models (VLMs) have made significant strides in image-text understanding tasks. However, their performance in 3D spatial comprehension, which is critical for embodied intelligence, remains limited. Recent advances have leveraged 3D point clouds and multi-view images as inputs, yielding promising results. However, we propose exploring a purely vision-based solution inspired by human perception, which merely relies on visual cues for 3D spatial understanding. This paper empirically investigates the limitations of VLMs in 3D spatial knowledge, revealing that the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.01428","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.01428/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.01428","created_at":"2026-07-05T10:28:31.630410+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.01428v4","created_at":"2026-07-05T10:28:31.630410+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.01428","created_at":"2026-07-05T10:28:31.630410+00:00"},{"alias_kind":"pith_short_12","alias_value":"3VDRO65GHSMF","created_at":"2026-07-05T10:28:31.630410+00:00"},{"alias_kind":"pith_short_16","alias_value":"3VDRO65GHSMFKOEJ","created_at":"2026-07-05T10:28:31.630410+00:00"},{"alias_kind":"pith_short_8","alias_value":"3VDRO65G","created_at":"2026-07-05T10:28:31.630410+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":28,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06534","citing_title":"CAIRN: Cross-Room 3D Scene Understanding with Topology-Aware Large Multimodal Models","ref_index":43,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24649","citing_title":"Agentic Collaborative Cognition for Zero-Shot 3D Understanding","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24649","citing_title":"Agentic Collaborative Cognition for Zero-Shot 3D Understanding","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19915","citing_title":"SpatialSV: Internalizing Interpretable 3D Spatial Awareness in MLLMs via Task-Oriented Visual Supervision","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19776","citing_title":"Occ-VLM: Occupancy Grounded Vision Language Model for Indoor Scene Understanding","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01784","citing_title":"SpaceEra++: A Unified Framework Towards 3D Spatial Reasoning in Video","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06891","citing_title":"Stream3D-VLM: Online 3D Spatial Understanding with Incremental Geometry Priors","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24456","citing_title":"EgoProx: Evaluating MLLMs on Egocentric 3D Proximity Reasoning Across a Cognitive Hierarchy","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25901","citing_title":"AgentGrounder: Zero-Shot 3D Visual Pointcloud Grounding using Multimodal Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00095","citing_title":"Bridging the 2D-3D Gap: A Hierarchical Semantic-Geometric Map for Vision Language Navigation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28060","citing_title":"ReScene: Structured Indoor Scene Reconstruction from Multi-View Captures","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28490","citing_title":"SSR3D-LLM: Structured Spatial Reasoning via Latent Steps for Fine-Grained Grounding in Unified 3D-LLMs","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30231","citing_title":"Beyond 3D VQAs: Injecting 3D Spatial Priors into Vision-Language Models for Enhanced Geometric Reasoning","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15876","citing_title":"Unlocking Dense Metric Depth Estimation in VLMs","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15876","citing_title":"Unlocking Dense Metric Depth Estimation in VLMs","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21471","citing_title":"SpatialBench: Benchmarking Multimodal Large Language Models for Spatial Cognition","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09965","citing_title":"Reinforcing Spatial Reasoning in Vision-Language Models with Interwoven Thinking and Visual Drawing","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08592","citing_title":"Boosting MLLM Spatial Reasoning with Geometrically Referenced 3D Scene Representations","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2603.17980","citing_title":"Feeling the Space: Egomotion-Aware Video Representation for Efficient and Accurate 3D Scene Understanding","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03296","citing_title":"3D-IDE: 3D Implicit Depth Emergent","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03318","citing_title":"EgoMind: Activating Spatial Cognition through Linguistic Reasoning in MLLMs","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02689","citing_title":"Efficient3D: A Unified Framework for Adaptive and Debiased Token Reduction in 3D MLLMs","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10106","citing_title":"ViSRA: A Video-based Spatial Reasoning Agent for Multi-modal Large Language Models","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3VDRO65GHSMFKOEJ4CG7NHO37B","json":"https://pith.science/pith/3VDRO65GHSMFKOEJ4CG7NHO37B.json","graph_json":"https://pith.science/api/pith-number/3VDRO65GHSMFKOEJ4CG7NHO37B/graph.json","events_json":"https://pith.science/api/pith-number/3VDRO65GHSMFKOEJ4CG7NHO37B/events.json","paper":"https://pith.science/paper/3VDRO65G"},"agent_actions":{"view_html":"https://pith.science/pith/3VDRO65GHSMFKOEJ4CG7NHO37B","download_json":"https://pith.science/pith/3VDRO65GHSMFKOEJ4CG7NHO37B.json","view_paper":"https://pith.science/paper/3VDRO65G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.01428&json=true","fetch_graph":"https://pith.science/api/pith-number/3VDRO65GHSMFKOEJ4CG7NHO37B/graph.json","fetch_events":"https://pith.science/api/pith-number/3VDRO65GHSMFKOEJ4CG7NHO37B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3VDRO65GHSMFKOEJ4CG7NHO37B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3VDRO65GHSMFKOEJ4CG7NHO37B/action/storage_attestation","attest_author":"https://pith.science/pith/3VDRO65GHSMFKOEJ4CG7NHO37B/action/author_attestation","sign_citation":"https://pith.science/pith/3VDRO65GHSMFKOEJ4CG7NHO37B/action/citation_signature","submit_replication":"https://pith.science/pith/3VDRO65GHSMFKOEJ4CG7NHO37B/action/replication_record"}},"created_at":"2026-07-05T10:28:31.630410+00:00","updated_at":"2026-07-05T10:28:31.630410+00:00"}