{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:L4CVDBAG4HQVRLB2TGZVVQMJ6H","short_pith_number":"pith:L4CVDBAG","schema_version":"1.0","canonical_sha256":"5f05518406e1e158ac3a99b35ac189f1f8cf858cdbe1be97b32c5294404c96df","source":{"kind":"arxiv","id":"2409.18125","version":3},"attestation_state":"computed","paper":{"title":"LLaVA-3D: A Simple yet Effective Pathway to Empowering LMMs with 3D-awareness","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenming Zhu, Jiangmiao Pang, Tai Wang, Wenwei Zhang, Xihui Liu","submitted_at":"2024-09-26T17:59:11Z","abstract_excerpt":"Recent advancements in Large Multimodal Models (LMMs) have greatly enhanced their proficiency in 2D visual understanding tasks, enabling them to effectively process and understand images and videos. However, the development of LMMs with 3D scene understanding capabilities has been hindered by the lack of large-scale 3D vision-language datasets and powerful 3D encoders. In this paper, we introduce a simple yet effective framework called LLaVA-3D. Leveraging the strong 2D visual understanding priors from LLaVA, our LLaVA-3D efficiently adapts LLaVA for 3D scene understanding without compromising"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.18125","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-09-26T17:59:11Z","cross_cats_sorted":[],"title_canon_sha256":"90c7fe5231073cde2d7e565a13f6aa2681b535881b79647451ee79ec01b5a457","abstract_canon_sha256":"cd63e23eaa72ddc6c3a5c4faf93b9ef9d64e764c18424bfe09ce1472a067d3e1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:54:36.982230Z","signature_b64":"RjXWc4HkIqRbj/2HMM5qttlQvbtxj4/zQU2w+SQku/k3+ctCxtJJg5SKjTYc51M9tJ0IMgPH3mKRlj3l1dNmDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5f05518406e1e158ac3a99b35ac189f1f8cf858cdbe1be97b32c5294404c96df","last_reissued_at":"2026-07-05T10:54:36.981759Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:54:36.981759Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLaVA-3D: A Simple yet Effective Pathway to Empowering LMMs with 3D-awareness","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenming Zhu, Jiangmiao Pang, Tai Wang, Wenwei Zhang, Xihui Liu","submitted_at":"2024-09-26T17:59:11Z","abstract_excerpt":"Recent advancements in Large Multimodal Models (LMMs) have greatly enhanced their proficiency in 2D visual understanding tasks, enabling them to effectively process and understand images and videos. However, the development of LMMs with 3D scene understanding capabilities has been hindered by the lack of large-scale 3D vision-language datasets and powerful 3D encoders. In this paper, we introduce a simple yet effective framework called LLaVA-3D. Leveraging the strong 2D visual understanding priors from LLaVA, our LLaVA-3D efficiently adapts LLaVA for 3D scene understanding without compromising"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.18125","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.18125/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.18125","created_at":"2026-07-05T10:54:36.981816+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.18125v3","created_at":"2026-07-05T10:54:36.981816+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.18125","created_at":"2026-07-05T10:54:36.981816+00:00"},{"alias_kind":"pith_short_12","alias_value":"L4CVDBAG4HQV","created_at":"2026-07-05T10:54:36.981816+00:00"},{"alias_kind":"pith_short_16","alias_value":"L4CVDBAG4HQVRLB2","created_at":"2026-07-05T10:54:36.981816+00:00"},{"alias_kind":"pith_short_8","alias_value":"L4CVDBAG","created_at":"2026-07-05T10:54:36.981816+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":31,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24068","citing_title":"ObsGraph: Hierarchical Observation Representation for Embodied Reasoning and Exploration","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22476","citing_title":"CVSBench: A Comprehensive Benchmark for Cross-view Spatial Reasoning and Dreaming","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19828","citing_title":"3D-PLOT-LLM: Part-Level Object Tokens for 3D Large Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19776","citing_title":"Occ-VLM: Occupancy Grounded Vision Language Model for Indoor Scene Understanding","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17539","citing_title":"Reinforcing Dual-Path Reasoning in Spatial Vision Language Models","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09669","citing_title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06476","citing_title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24642","citing_title":"Understanding the Impact of Geometric Foundation Models on Vision-Language-Action Models","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26230","citing_title":"Geometry-Aware Representation Denoising for Robust Multi-view 3D Reconstruction","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25326","citing_title":"Perceive-then-Plan: Layout-as-Policy for Monocular 3D Scene Layout Estimation","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30231","citing_title":"Beyond 3D VQAs: Injecting 3D Spatial Priors into Vision-Language Models for Enhanced Geometric Reasoning","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2512.10719","citing_title":"SpaceDrive: Infusing Spatial Awareness into VLM-based Autonomous Driving","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14171","citing_title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15876","citing_title":"Unlocking Dense Metric Depth Estimation in VLMs","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15876","citing_title":"Unlocking Dense Metric Depth Estimation in VLMs","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20165","citing_title":"CaMo: Camera Motion Grounded Evaluation and Training for Vision-Language Models","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2511.16567","citing_title":"POMA-3D: The Point Map Way to 3D Scene Understanding","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2511.19972","citing_title":"Boosting Reasoning in Large Multimodal Models via Activation Replay","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21471","citing_title":"SpatialBench: Benchmarking Multimodal Large Language Models for Spatial Cognition","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01400","citing_title":"Token Reduction via Local and Global Contexts Optimization for Efficient Video Large Language Models","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08592","citing_title":"Boosting MLLM Spatial Reasoning with Geometrically Referenced 3D Scene Representations","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2603.17980","citing_title":"Feeling the Space: Egomotion-Aware Video Representation for Efficient and Accurate 3D Scene Understanding","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2603.23404","citing_title":"Unleashing Spatial Reasoning in Multimodal Large Language Models via Textual Representation Guided Reasoning","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H","json":"https://pith.science/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H.json","graph_json":"https://pith.science/api/pith-number/L4CVDBAG4HQVRLB2TGZVVQMJ6H/graph.json","events_json":"https://pith.science/api/pith-number/L4CVDBAG4HQVRLB2TGZVVQMJ6H/events.json","paper":"https://pith.science/paper/L4CVDBAG"},"agent_actions":{"view_html":"https://pith.science/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H","download_json":"https://pith.science/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H.json","view_paper":"https://pith.science/paper/L4CVDBAG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.18125&json=true","fetch_graph":"https://pith.science/api/pith-number/L4CVDBAG4HQVRLB2TGZVVQMJ6H/graph.json","fetch_events":"https://pith.science/api/pith-number/L4CVDBAG4HQVRLB2TGZVVQMJ6H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H/action/storage_attestation","attest_author":"https://pith.science/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H/action/author_attestation","sign_citation":"https://pith.science/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H/action/citation_signature","submit_replication":"https://pith.science/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H/action/replication_record"}},"created_at":"2026-07-05T10:54:36.981816+00:00","updated_at":"2026-07-05T10:54:36.981816+00:00"}