{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:L4CVDBAG4HQVRLB2TGZVVQMJ6H","short_pith_number":"pith:L4CVDBAG","canonical_record":{"source":{"id":"2409.18125","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-09-26T17:59:11Z","cross_cats_sorted":[],"title_canon_sha256":"90c7fe5231073cde2d7e565a13f6aa2681b535881b79647451ee79ec01b5a457","abstract_canon_sha256":"cd63e23eaa72ddc6c3a5c4faf93b9ef9d64e764c18424bfe09ce1472a067d3e1"},"schema_version":"1.0"},"canonical_sha256":"5f05518406e1e158ac3a99b35ac189f1f8cf858cdbe1be97b32c5294404c96df","source":{"kind":"arxiv","id":"2409.18125","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2409.18125","created_at":"2026-07-05T10:54:36Z"},{"alias_kind":"arxiv_version","alias_value":"2409.18125v3","created_at":"2026-07-05T10:54:36Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.18125","created_at":"2026-07-05T10:54:36Z"},{"alias_kind":"pith_short_12","alias_value":"L4CVDBAG4HQV","created_at":"2026-07-05T10:54:36Z"},{"alias_kind":"pith_short_16","alias_value":"L4CVDBAG4HQVRLB2","created_at":"2026-07-05T10:54:36Z"},{"alias_kind":"pith_short_8","alias_value":"L4CVDBAG","created_at":"2026-07-05T10:54:36Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:L4CVDBAG4HQVRLB2TGZVVQMJ6H","target":"record","payload":{"canonical_record":{"source":{"id":"2409.18125","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-09-26T17:59:11Z","cross_cats_sorted":[],"title_canon_sha256":"90c7fe5231073cde2d7e565a13f6aa2681b535881b79647451ee79ec01b5a457","abstract_canon_sha256":"cd63e23eaa72ddc6c3a5c4faf93b9ef9d64e764c18424bfe09ce1472a067d3e1"},"schema_version":"1.0"},"canonical_sha256":"5f05518406e1e158ac3a99b35ac189f1f8cf858cdbe1be97b32c5294404c96df","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:54:36.982230Z","signature_b64":"RjXWc4HkIqRbj/2HMM5qttlQvbtxj4/zQU2w+SQku/k3+ctCxtJJg5SKjTYc51M9tJ0IMgPH3mKRlj3l1dNmDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5f05518406e1e158ac3a99b35ac189f1f8cf858cdbe1be97b32c5294404c96df","last_reissued_at":"2026-07-05T10:54:36.981759Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:54:36.981759Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2409.18125","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:54:36Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"kCrJ07HFW58FQFBIXJUmma32HtxWe/yfAQy+WXPoNU356r+wwxLi6R7z9XRd1fTeul6wuN4j7jhZQPAFTTeDBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T16:37:22.981460Z"},"content_sha256":"5a1e6256b7e5829974fcb1ad6a46dc47242e4548004d49c6f1294e99907677e4","schema_version":"1.0","event_id":"sha256:5a1e6256b7e5829974fcb1ad6a46dc47242e4548004d49c6f1294e99907677e4"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:L4CVDBAG4HQVRLB2TGZVVQMJ6H","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"LLaVA-3D: A Simple yet Effective Pathway to Empowering LMMs with 3D-awareness","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenming Zhu, Jiangmiao Pang, Tai Wang, Wenwei Zhang, Xihui Liu","submitted_at":"2024-09-26T17:59:11Z","abstract_excerpt":"Recent advancements in Large Multimodal Models (LMMs) have greatly enhanced their proficiency in 2D visual understanding tasks, enabling them to effectively process and understand images and videos. However, the development of LMMs with 3D scene understanding capabilities has been hindered by the lack of large-scale 3D vision-language datasets and powerful 3D encoders. In this paper, we introduce a simple yet effective framework called LLaVA-3D. Leveraging the strong 2D visual understanding priors from LLaVA, our LLaVA-3D efficiently adapts LLaVA for 3D scene understanding without compromising"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.18125","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.18125/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:54:36Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UNDVPjt86fqQSN58zvJEFumc21E6Oxx8yNoVdO6oi3H5CKraXYDi3qRcT6TQpJktAzB6Qc78pzgWhmT+B3TmDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T16:37:22.981961Z"},"content_sha256":"63df2eefe1534c751e47cede8778bc49844a0dc46c30848d09d5d50c0fd36503","schema_version":"1.0","event_id":"sha256:63df2eefe1534c751e47cede8778bc49844a0dc46c30848d09d5d50c0fd36503"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H/bundle.json","state_url":"https://pith.science/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T16:37:22Z","links":{"resolver":"https://pith.science/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H","bundle":"https://pith.science/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H/bundle.json","state":"https://pith.science/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H/state.json","well_known_bundle":"https://pith.science/.well-known/pith/L4CVDBAG4HQVRLB2TGZVVQMJ6H/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:L4CVDBAG4HQVRLB2TGZVVQMJ6H","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"cd63e23eaa72ddc6c3a5c4faf93b9ef9d64e764c18424bfe09ce1472a067d3e1","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-09-26T17:59:11Z","title_canon_sha256":"90c7fe5231073cde2d7e565a13f6aa2681b535881b79647451ee79ec01b5a457"},"schema_version":"1.0","source":{"id":"2409.18125","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2409.18125","created_at":"2026-07-05T10:54:36Z"},{"alias_kind":"arxiv_version","alias_value":"2409.18125v3","created_at":"2026-07-05T10:54:36Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.18125","created_at":"2026-07-05T10:54:36Z"},{"alias_kind":"pith_short_12","alias_value":"L4CVDBAG4HQV","created_at":"2026-07-05T10:54:36Z"},{"alias_kind":"pith_short_16","alias_value":"L4CVDBAG4HQVRLB2","created_at":"2026-07-05T10:54:36Z"},{"alias_kind":"pith_short_8","alias_value":"L4CVDBAG","created_at":"2026-07-05T10:54:36Z"}],"graph_snapshots":[{"event_id":"sha256:63df2eefe1534c751e47cede8778bc49844a0dc46c30848d09d5d50c0fd36503","target":"graph","created_at":"2026-07-05T10:54:36Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2409.18125/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Recent advancements in Large Multimodal Models (LMMs) have greatly enhanced their proficiency in 2D visual understanding tasks, enabling them to effectively process and understand images and videos. However, the development of LMMs with 3D scene understanding capabilities has been hindered by the lack of large-scale 3D vision-language datasets and powerful 3D encoders. In this paper, we introduce a simple yet effective framework called LLaVA-3D. Leveraging the strong 2D visual understanding priors from LLaVA, our LLaVA-3D efficiently adapts LLaVA for 3D scene understanding without compromising","authors_text":"Chenming Zhu, Jiangmiao Pang, Tai Wang, Wenwei Zhang, Xihui Liu","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-09-26T17:59:11Z","title":"LLaVA-3D: A Simple yet Effective Pathway to Empowering LMMs with 3D-awareness"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.18125","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:5a1e6256b7e5829974fcb1ad6a46dc47242e4548004d49c6f1294e99907677e4","target":"record","created_at":"2026-07-05T10:54:36Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"cd63e23eaa72ddc6c3a5c4faf93b9ef9d64e764c18424bfe09ce1472a067d3e1","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-09-26T17:59:11Z","title_canon_sha256":"90c7fe5231073cde2d7e565a13f6aa2681b535881b79647451ee79ec01b5a457"},"schema_version":"1.0","source":{"id":"2409.18125","kind":"arxiv","version":3}},"canonical_sha256":"5f05518406e1e158ac3a99b35ac189f1f8cf858cdbe1be97b32c5294404c96df","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"5f05518406e1e158ac3a99b35ac189f1f8cf858cdbe1be97b32c5294404c96df","first_computed_at":"2026-07-05T10:54:36.981759Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:54:36.981759Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"RjXWc4HkIqRbj/2HMM5qttlQvbtxj4/zQU2w+SQku/k3+ctCxtJJg5SKjTYc51M9tJ0IMgPH3mKRlj3l1dNmDA==","signature_status":"signed_v1","signed_at":"2026-07-05T10:54:36.982230Z","signed_message":"canonical_sha256_bytes"},"source_id":"2409.18125","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:5a1e6256b7e5829974fcb1ad6a46dc47242e4548004d49c6f1294e99907677e4","sha256:63df2eefe1534c751e47cede8778bc49844a0dc46c30848d09d5d50c0fd36503"],"state_sha256":"7b9b70a20e80a62e32132c1b044e94f1764536dbfea66d4af11a8d17e1adfd05"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"/vu6VLxEr4ty2kOSX1h+SN/HoJAPy4iCAbRw+5CXJfYNo6Xd3h/mOqJAPvDXFC29ReXA7V+NeoNIyHAZ+KiADA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T16:37:22.985828Z","bundle_sha256":"2d1d9fdaa849fdb539cbbed97fe83111e05a2dc491d2f25c3b98ebd7608e2446"}}