{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4Z32NCHWXAXWLMJTZO2VRNZ4ZR","short_pith_number":"pith:4Z32NCHW","schema_version":"1.0","canonical_sha256":"e677a688f6b82f65b133cbb558b73ccc711b7e2e712d5e50783fd4302f9fe9a3","source":{"kind":"arxiv","id":"2401.11395","version":3},"attestation_state":"computed","paper":{"title":"UniM-OV3D: Uni-Modality Open-Vocabulary 3D Scene Understanding with Fine-Grained Feature Representation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengjie Wang, Jiangning Zhang, Jinlong Peng, Kai Wu, Mingang Chen, Qingdong He, Xiaozhong Ji, Yabiao Wang, Yunsheng Wu, Zhengkai Jiang","submitted_at":"2024-01-21T04:13:58Z","abstract_excerpt":"3D open-vocabulary scene understanding aims to recognize arbitrary novel categories beyond the base label space. However, existing works not only fail to fully utilize all the available modal information in the 3D domain but also lack sufficient granularity in representing the features of each modality. In this paper, we propose a unified multimodal 3D open-vocabulary scene understanding network, namely UniM-OV3D, which aligns point clouds with image, language and depth. To better integrate global and local features of the point clouds, we design a hierarchical point cloud feature extraction m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.11395","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-21T04:13:58Z","cross_cats_sorted":[],"title_canon_sha256":"74128c21e2ec2dbf35bd20c7fe0f3e6215aa514e1c9c0f320c6fe89ae6fb7432","abstract_canon_sha256":"b34371f135f86efdf53b6bb3edef17f8f41787ffa216d123ea479254dac0a102"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:10:22.481401Z","signature_b64":"IFMiL/uXWMycCxnV+iEoJr8BxRZM2/zlyxz/zgHvDRwU16bwfxM6AYIdKe1MwhaeBmjxM2v9Q0+VHLkqMPOLCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e677a688f6b82f65b133cbb558b73ccc711b7e2e712d5e50783fd4302f9fe9a3","last_reissued_at":"2026-07-05T08:10:22.480921Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:10:22.480921Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"UniM-OV3D: Uni-Modality Open-Vocabulary 3D Scene Understanding with Fine-Grained Feature Representation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengjie Wang, Jiangning Zhang, Jinlong Peng, Kai Wu, Mingang Chen, Qingdong He, Xiaozhong Ji, Yabiao Wang, Yunsheng Wu, Zhengkai Jiang","submitted_at":"2024-01-21T04:13:58Z","abstract_excerpt":"3D open-vocabulary scene understanding aims to recognize arbitrary novel categories beyond the base label space. However, existing works not only fail to fully utilize all the available modal information in the 3D domain but also lack sufficient granularity in representing the features of each modality. In this paper, we propose a unified multimodal 3D open-vocabulary scene understanding network, namely UniM-OV3D, which aligns point clouds with image, language and depth. To better integrate global and local features of the point clouds, we design a hierarchical point cloud feature extraction m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.11395","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.11395/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.11395","created_at":"2026-07-05T08:10:22.480977+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.11395v3","created_at":"2026-07-05T08:10:22.480977+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.11395","created_at":"2026-07-05T08:10:22.480977+00:00"},{"alias_kind":"pith_short_12","alias_value":"4Z32NCHWXAXW","created_at":"2026-07-05T08:10:22.480977+00:00"},{"alias_kind":"pith_short_16","alias_value":"4Z32NCHWXAXWLMJT","created_at":"2026-07-05T08:10:22.480977+00:00"},{"alias_kind":"pith_short_8","alias_value":"4Z32NCHW","created_at":"2026-07-05T08:10:22.480977+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.03370","citing_title":"ShelfGaussian: Shelf-Supervised Open-Vocabulary Gaussian-based 3D Scene Understanding","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4Z32NCHWXAXWLMJTZO2VRNZ4ZR","json":"https://pith.science/pith/4Z32NCHWXAXWLMJTZO2VRNZ4ZR.json","graph_json":"https://pith.science/api/pith-number/4Z32NCHWXAXWLMJTZO2VRNZ4ZR/graph.json","events_json":"https://pith.science/api/pith-number/4Z32NCHWXAXWLMJTZO2VRNZ4ZR/events.json","paper":"https://pith.science/paper/4Z32NCHW"},"agent_actions":{"view_html":"https://pith.science/pith/4Z32NCHWXAXWLMJTZO2VRNZ4ZR","download_json":"https://pith.science/pith/4Z32NCHWXAXWLMJTZO2VRNZ4ZR.json","view_paper":"https://pith.science/paper/4Z32NCHW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.11395&json=true","fetch_graph":"https://pith.science/api/pith-number/4Z32NCHWXAXWLMJTZO2VRNZ4ZR/graph.json","fetch_events":"https://pith.science/api/pith-number/4Z32NCHWXAXWLMJTZO2VRNZ4ZR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4Z32NCHWXAXWLMJTZO2VRNZ4ZR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4Z32NCHWXAXWLMJTZO2VRNZ4ZR/action/storage_attestation","attest_author":"https://pith.science/pith/4Z32NCHWXAXWLMJTZO2VRNZ4ZR/action/author_attestation","sign_citation":"https://pith.science/pith/4Z32NCHWXAXWLMJTZO2VRNZ4ZR/action/citation_signature","submit_replication":"https://pith.science/pith/4Z32NCHWXAXWLMJTZO2VRNZ4ZR/action/replication_record"}},"created_at":"2026-07-05T08:10:22.480977+00:00","updated_at":"2026-07-05T08:10:22.480977+00:00"}