{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:D7XNAAEZVPH5UTBVIAJKAQMHMN","short_pith_number":"pith:D7XNAAEZ","schema_version":"1.0","canonical_sha256":"1feed00099abcfda4c354012a04187636ca258ba76a02420abece504f82aebd0","source":{"kind":"arxiv","id":"2412.00493","version":2},"attestation_state":"computed","paper":{"title":"Video-3D LLM: Learning Position-Aware Video Representation for 3D Scene Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Duo Zheng, Liwei Wang, Shijia Huang","submitted_at":"2024-11-30T14:28:53Z","abstract_excerpt":"The rapid advancement of Multimodal Large Language Models (MLLMs) has significantly impacted various multimodal tasks. However, these models face challenges in tasks that require spatial understanding within 3D environments. Efforts to enhance MLLMs, such as incorporating point cloud features, have been made, yet a considerable gap remains between the models' learned representations and the inherent complexity of 3D scenes. This discrepancy largely stems from the training of MLLMs on predominantly 2D data, which restricts their effectiveness in comprehending 3D spaces. To address this issue, i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.00493","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-30T14:28:53Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"d0b563db452e60a938decd41ec979210de678843e9812de2e7163b6fa7289cf3","abstract_canon_sha256":"84bb60c384a267417d0a2082bc5f78356a294afeb8beded7688979732df7fdb8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:40:14.672431Z","signature_b64":"Ge+6ZcWVSXkG101uG2MFkdJdWmKprjo0fEJxMCcmjM+a/QQCtBOpzPYmmm5lED0u7oigm1Qa5k4joeqdF2dPAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1feed00099abcfda4c354012a04187636ca258ba76a02420abece504f82aebd0","last_reissued_at":"2026-07-05T10:40:14.671989Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:40:14.671989Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Video-3D LLM: Learning Position-Aware Video Representation for 3D Scene Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Duo Zheng, Liwei Wang, Shijia Huang","submitted_at":"2024-11-30T14:28:53Z","abstract_excerpt":"The rapid advancement of Multimodal Large Language Models (MLLMs) has significantly impacted various multimodal tasks. However, these models face challenges in tasks that require spatial understanding within 3D environments. Efforts to enhance MLLMs, such as incorporating point cloud features, have been made, yet a considerable gap remains between the models' learned representations and the inherent complexity of 3D scenes. This discrepancy largely stems from the training of MLLMs on predominantly 2D data, which restricts their effectiveness in comprehending 3D spaces. To address this issue, i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.00493","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.00493/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.00493","created_at":"2026-07-05T10:40:14.672046+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.00493v2","created_at":"2026-07-05T10:40:14.672046+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.00493","created_at":"2026-07-05T10:40:14.672046+00:00"},{"alias_kind":"pith_short_12","alias_value":"D7XNAAEZVPH5","created_at":"2026-07-05T10:40:14.672046+00:00"},{"alias_kind":"pith_short_16","alias_value":"D7XNAAEZVPH5UTBV","created_at":"2026-07-05T10:40:14.672046+00:00"},{"alias_kind":"pith_short_8","alias_value":"D7XNAAEZ","created_at":"2026-07-05T10:40:14.672046+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10506","citing_title":"A Progressive Training Strategy for Vision-Language Models to Counteract Spatio-Temporal Hallucinations in Embodied Reasoning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09167","citing_title":"MAG-3D: Multi-Agent Grounded Reasoning for 3D Understanding","ref_index":54,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D7XNAAEZVPH5UTBVIAJKAQMHMN","json":"https://pith.science/pith/D7XNAAEZVPH5UTBVIAJKAQMHMN.json","graph_json":"https://pith.science/api/pith-number/D7XNAAEZVPH5UTBVIAJKAQMHMN/graph.json","events_json":"https://pith.science/api/pith-number/D7XNAAEZVPH5UTBVIAJKAQMHMN/events.json","paper":"https://pith.science/paper/D7XNAAEZ"},"agent_actions":{"view_html":"https://pith.science/pith/D7XNAAEZVPH5UTBVIAJKAQMHMN","download_json":"https://pith.science/pith/D7XNAAEZVPH5UTBVIAJKAQMHMN.json","view_paper":"https://pith.science/paper/D7XNAAEZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.00493&json=true","fetch_graph":"https://pith.science/api/pith-number/D7XNAAEZVPH5UTBVIAJKAQMHMN/graph.json","fetch_events":"https://pith.science/api/pith-number/D7XNAAEZVPH5UTBVIAJKAQMHMN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D7XNAAEZVPH5UTBVIAJKAQMHMN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D7XNAAEZVPH5UTBVIAJKAQMHMN/action/storage_attestation","attest_author":"https://pith.science/pith/D7XNAAEZVPH5UTBVIAJKAQMHMN/action/author_attestation","sign_citation":"https://pith.science/pith/D7XNAAEZVPH5UTBVIAJKAQMHMN/action/citation_signature","submit_replication":"https://pith.science/pith/D7XNAAEZVPH5UTBVIAJKAQMHMN/action/replication_record"}},"created_at":"2026-07-05T10:40:14.672046+00:00","updated_at":"2026-07-05T10:40:14.672046+00:00"}