{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:X3WBVDHPCTM5JGX4O3PJ6HODDO","short_pith_number":"pith:X3WBVDHP","schema_version":"1.0","canonical_sha256":"beec1a8cef14d9d49afc76de9f1dc31b82951a783fc1c137e28de8ca8a349510","source":{"kind":"arxiv","id":"2403.11401","version":2},"attestation_state":"computed","paper":{"title":"Scene-LLM: Extending Language Model for 3D Visual Understanding and Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jingyu Liu, Rao Fu, Wenhan Xiong, Xilun Chen, Yixin Nie","submitted_at":"2024-03-18T01:18:48Z","abstract_excerpt":"This paper introduces Scene-LLM, a 3D-visual-language model that enhances embodied agents' abilities in interactive 3D indoor environments by integrating the reasoning strengths of Large Language Models (LLMs). Scene-LLM adopts a hybrid 3D visual feature representation, that incorporates dense spatial information and supports scene state updates. The model employs a projection layer to efficiently project these features in the pre-trained textual embedding space, enabling effective interpretation of 3D visual information. Unique to our approach is the integration of both scene-level and ego-ce"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.11401","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-18T01:18:48Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4a0efff5197fd08e05de67322c5678b547b675d6a10f07782ebcb5179589759a","abstract_canon_sha256":"237b30d39cc28259f4cbd0bcc7fe8ad5edb498ebc1f632c0c671936e615e9bd0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:59:36.316575Z","signature_b64":"DVhDFrIC83W3kLci/A0BsOyTD0T4DvpKo15gYaopJIg06Q5EltDnJhTO5jais/e6yKX+xuWennjvexRWTlx0Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"beec1a8cef14d9d49afc76de9f1dc31b82951a783fc1c137e28de8ca8a349510","last_reissued_at":"2026-07-05T07:59:36.315952Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:59:36.315952Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scene-LLM: Extending Language Model for 3D Visual Understanding and Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jingyu Liu, Rao Fu, Wenhan Xiong, Xilun Chen, Yixin Nie","submitted_at":"2024-03-18T01:18:48Z","abstract_excerpt":"This paper introduces Scene-LLM, a 3D-visual-language model that enhances embodied agents' abilities in interactive 3D indoor environments by integrating the reasoning strengths of Large Language Models (LLMs). Scene-LLM adopts a hybrid 3D visual feature representation, that incorporates dense spatial information and supports scene state updates. The model employs a projection layer to efficiently project these features in the pre-trained textual embedding space, enabling effective interpretation of 3D visual information. Unique to our approach is the integration of both scene-level and ego-ce"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.11401","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.11401/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.11401","created_at":"2026-07-05T07:59:36.316067+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.11401v2","created_at":"2026-07-05T07:59:36.316067+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.11401","created_at":"2026-07-05T07:59:36.316067+00:00"},{"alias_kind":"pith_short_12","alias_value":"X3WBVDHPCTM5","created_at":"2026-07-05T07:59:36.316067+00:00"},{"alias_kind":"pith_short_16","alias_value":"X3WBVDHPCTM5JGX4","created_at":"2026-07-05T07:59:36.316067+00:00"},{"alias_kind":"pith_short_8","alias_value":"X3WBVDHP","created_at":"2026-07-05T07:59:36.316067+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":25,"internal_anchor_count":3,"sample":[{"citing_arxiv_id":"2607.07001","citing_title":"Ego-Human Motion Prediction with 3D-Aware LLM","ref_index":25,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06534","citing_title":"CAIRN: Cross-Room 3D Scene Understanding with Topology-Aware Large Multimodal Models","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06565","citing_title":"ELSA3D: Elastic Semantic Anchoring for Unified 3D Understanding and Generation","ref_index":105,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19776","citing_title":"Occ-VLM: Occupancy Grounded Vision Language Model for Indoor Scene Understanding","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06485","citing_title":"PAR3D: A Unified 3D-MLLM with Part-Aware Representation for Scene Understanding","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28081","citing_title":"Context-Aware Explanations for Spatialized Document Layouts","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24321","citing_title":"Unified 3D Scene Understanding Through Physical World Modeling","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24456","citing_title":"EgoProx: Evaluating MLLMs on Egocentric 3D Proximity Reasoning Across a Cognitive Hierarchy","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26239","citing_title":"Sentinel: Embodied Cooperative Spatial Reasoning and Planning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30231","citing_title":"Beyond 3D VQAs: Injecting 3D Spatial Priors into Vision-Language Models for Enhanced Geometric Reasoning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2512.10719","citing_title":"SpaceDrive: Infusing Spatial Awareness into VLM-based Autonomous Driving","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20837","citing_title":"ArchSIBench: Benchmarking the Architectural Spatial Intelligence of Vision-Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17118","citing_title":"Differentiable Optimization Layers for Guaranteed Fairness in Deep Learning","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2511.16567","citing_title":"POMA-3D: The Point Map Way to 3D Scene Understanding","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21471","citing_title":"SpatialBench: Benchmarking Multimodal Large Language Models for Spatial Cognition","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23365","citing_title":"SpatialMosaic: A Multiview VLM Dataset for Partial Visibility","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08592","citing_title":"Boosting MLLM Spatial Reasoning with Geometrically Referenced 3D Scene Representations","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27507","citing_title":"Chat-Scene++: Exploiting Context-Rich Object Identification for 3D LLM","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2501.15830","citing_title":"SpatialVLA: Exploring Spatial Representations for Visual-Language-Action Model","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09218","citing_title":"Flame3D: Zero-shot Compositional Reasoning of 3D Scenes with Agentic Language Models","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19509","citing_title":"Assessing VLM-Driven Semantic-Affordance Inference for Non-Humanoid Robot Morphologies","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08645","citing_title":"3D-VCD: Hallucination Mitigation in 3D-LLM Embodied Agents through Visual Contrastive Decoding","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21160","citing_title":"Reinforcing 3D Understanding in Point-VLMs via Geometric Reward Credit Assignment","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X3WBVDHPCTM5JGX4O3PJ6HODDO","json":"https://pith.science/pith/X3WBVDHPCTM5JGX4O3PJ6HODDO.json","graph_json":"https://pith.science/api/pith-number/X3WBVDHPCTM5JGX4O3PJ6HODDO/graph.json","events_json":"https://pith.science/api/pith-number/X3WBVDHPCTM5JGX4O3PJ6HODDO/events.json","paper":"https://pith.science/paper/X3WBVDHP"},"agent_actions":{"view_html":"https://pith.science/pith/X3WBVDHPCTM5JGX4O3PJ6HODDO","download_json":"https://pith.science/pith/X3WBVDHPCTM5JGX4O3PJ6HODDO.json","view_paper":"https://pith.science/paper/X3WBVDHP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.11401&json=true","fetch_graph":"https://pith.science/api/pith-number/X3WBVDHPCTM5JGX4O3PJ6HODDO/graph.json","fetch_events":"https://pith.science/api/pith-number/X3WBVDHPCTM5JGX4O3PJ6HODDO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X3WBVDHPCTM5JGX4O3PJ6HODDO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X3WBVDHPCTM5JGX4O3PJ6HODDO/action/storage_attestation","attest_author":"https://pith.science/pith/X3WBVDHPCTM5JGX4O3PJ6HODDO/action/author_attestation","sign_citation":"https://pith.science/pith/X3WBVDHPCTM5JGX4O3PJ6HODDO/action/citation_signature","submit_replication":"https://pith.science/pith/X3WBVDHPCTM5JGX4O3PJ6HODDO/action/replication_record"}},"created_at":"2026-07-05T07:59:36.316067+00:00","updated_at":"2026-07-05T07:59:36.316067+00:00"}