{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:37AJNUXFPXPDWHLXVVZV4YBYTS","short_pith_number":"pith:37AJNUXF","schema_version":"1.0","canonical_sha256":"dfc096d2e57dde3b1d77ad735e60389ca742d03c9b8a652c5936873beec042a0","source":{"kind":"arxiv","id":"2308.08769","version":1},"attestation_state":"computed","paper":{"title":"Chat-3D: Data-efficiently Tuning Large Language Model for Universal Dialogue of 3D Scenes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haifeng Huang, Yang Zhao, Zehan Wang, Zhou Zhao, Ziang Zhang","submitted_at":"2023-08-17T03:52:15Z","abstract_excerpt":"3D scene understanding has gained significant attention due to its wide range of applications. However, existing methods for 3D scene understanding are limited to specific downstream tasks, which hinders their practicality in real-world applications. This paper presents Chat-3D, which combines the 3D visual perceptual ability of pre-trained 3D representations and the impressive reasoning and conversation capabilities of advanced LLMs to achieve the first universal dialogue systems for 3D scenes. Specifically, we align 3D representations into the feature space of LLMs, thus enabling LLMs to per"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.08769","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-08-17T03:52:15Z","cross_cats_sorted":[],"title_canon_sha256":"73f03d92806ed5a674e7fd97d12d6c5d6f119fded594df9de0a5c0850d86d664","abstract_canon_sha256":"9c7a86ece3e4482c8a027410c4291ec987fa60359404d5be2d7dc75abf97b255"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:42:11.038123Z","signature_b64":"MYLIH/dV1cLYyg88ynQ1RR8oUTDWnDH7MxRvFgFsl3zt+KZbr119sYOLlh4AYDan8ZkNGuNi2EJAdwaey2XTAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dfc096d2e57dde3b1d77ad735e60389ca742d03c9b8a652c5936873beec042a0","last_reissued_at":"2026-07-05T06:42:11.037783Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:42:11.037783Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Chat-3D: Data-efficiently Tuning Large Language Model for Universal Dialogue of 3D Scenes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haifeng Huang, Yang Zhao, Zehan Wang, Zhou Zhao, Ziang Zhang","submitted_at":"2023-08-17T03:52:15Z","abstract_excerpt":"3D scene understanding has gained significant attention due to its wide range of applications. However, existing methods for 3D scene understanding are limited to specific downstream tasks, which hinders their practicality in real-world applications. This paper presents Chat-3D, which combines the 3D visual perceptual ability of pre-trained 3D representations and the impressive reasoning and conversation capabilities of advanced LLMs to achieve the first universal dialogue systems for 3D scenes. Specifically, we align 3D representations into the feature space of LLMs, thus enabling LLMs to per"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.08769","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.08769/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.08769","created_at":"2026-07-05T06:42:11.037836+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.08769v1","created_at":"2026-07-05T06:42:11.037836+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.08769","created_at":"2026-07-05T06:42:11.037836+00:00"},{"alias_kind":"pith_short_12","alias_value":"37AJNUXFPXPD","created_at":"2026-07-05T06:42:11.037836+00:00"},{"alias_kind":"pith_short_16","alias_value":"37AJNUXFPXPDWHLX","created_at":"2026-07-05T06:42:11.037836+00:00"},{"alias_kind":"pith_short_8","alias_value":"37AJNUXF","created_at":"2026-07-05T06:42:11.037836+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06565","citing_title":"ELSA3D: Elastic Semantic Anchoring for Unified 3D Understanding and Generation","ref_index":73,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22476","citing_title":"CVSBench: A Comprehensive Benchmark for Cross-view Spatial Reasoning and Dreaming","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19776","citing_title":"Occ-VLM: Occupancy Grounded Vision Language Model for Indoor Scene Understanding","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19253","citing_title":"OneCanvas: 3D Scene Understanding via Panoramic Reprojection","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24456","citing_title":"EgoProx: Evaluating MLLMs on Egocentric 3D Proximity Reasoning Across a Cognitive Hierarchy","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30231","citing_title":"Beyond 3D VQAs: Injecting 3D Spatial Priors into Vision-Language Models for Enhanced Geometric Reasoning","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19528","citing_title":"Towards Camera-Robust 3D Localization: Equation-Anchored Tool-Use for MLLMs","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2511.16567","citing_title":"POMA-3D: The Point Map Way to 3D Scene Understanding","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08592","citing_title":"Boosting MLLM Spatial Reasoning with Geometrically Referenced 3D Scene Representations","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2603.17980","citing_title":"Feeling the Space: Egomotion-Aware Video Representation for Efficient and Accurate 3D Scene Understanding","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03296","citing_title":"3D-IDE: 3D Implicit Depth Emergent","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27507","citing_title":"Chat-Scene++: Exploiting Context-Rich Object Identification for 3D LLM","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03318","citing_title":"EgoMind: Activating Spatial Cognition through Linguistic Reasoning in MLLMs","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01736","citing_title":"Multi-Scale Gaussian-Language Map for Zero-shot Embodied Navigation and Reasoning","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21160","citing_title":"Reinforcing 3D Understanding in Point-VLMs via Geometric Reward Credit Assignment","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/37AJNUXFPXPDWHLXVVZV4YBYTS","json":"https://pith.science/pith/37AJNUXFPXPDWHLXVVZV4YBYTS.json","graph_json":"https://pith.science/api/pith-number/37AJNUXFPXPDWHLXVVZV4YBYTS/graph.json","events_json":"https://pith.science/api/pith-number/37AJNUXFPXPDWHLXVVZV4YBYTS/events.json","paper":"https://pith.science/paper/37AJNUXF"},"agent_actions":{"view_html":"https://pith.science/pith/37AJNUXFPXPDWHLXVVZV4YBYTS","download_json":"https://pith.science/pith/37AJNUXFPXPDWHLXVVZV4YBYTS.json","view_paper":"https://pith.science/paper/37AJNUXF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.08769&json=true","fetch_graph":"https://pith.science/api/pith-number/37AJNUXFPXPDWHLXVVZV4YBYTS/graph.json","fetch_events":"https://pith.science/api/pith-number/37AJNUXFPXPDWHLXVVZV4YBYTS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/37AJNUXFPXPDWHLXVVZV4YBYTS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/37AJNUXFPXPDWHLXVVZV4YBYTS/action/storage_attestation","attest_author":"https://pith.science/pith/37AJNUXFPXPDWHLXVVZV4YBYTS/action/author_attestation","sign_citation":"https://pith.science/pith/37AJNUXFPXPDWHLXVVZV4YBYTS/action/citation_signature","submit_replication":"https://pith.science/pith/37AJNUXFPXPDWHLXVVZV4YBYTS/action/replication_record"}},"created_at":"2026-07-05T06:42:11.037836+00:00","updated_at":"2026-07-05T06:42:11.037836+00:00"}