{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YZRBQDE4QLJUANZAKKPFQDXJ7U","short_pith_number":"pith:YZRBQDE4","schema_version":"1.0","canonical_sha256":"c662180c9c82d3403720529e580ee9fd1abf5571130acbc082094fe152aa6bd8","source":{"kind":"arxiv","id":"2410.24001","version":1},"attestation_state":"computed","paper":{"title":"ImOV3D: Learning Open-Vocabulary Point Clouds 3D Object Detection from Only 2D Images","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Li Yi, Timing Yang, Yuanliang Ju","submitted_at":"2024-10-31T15:02:05Z","abstract_excerpt":"Open-vocabulary 3D object detection (OV-3Det) aims to generalize beyond the limited number of base categories labeled during the training phase. The biggest bottleneck is the scarcity of annotated 3D data, whereas 2D image datasets are abundant and richly annotated. Consequently, it is intuitive to leverage the wealth of annotations in 2D images to alleviate the inherent data scarcity in OV-3Det. In this paper, we push the task setup to its limits by exploring the potential of using solely 2D images to learn OV-3Det. The major challenges for this setup is the modality gap between training imag"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.24001","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-31T15:02:05Z","cross_cats_sorted":[],"title_canon_sha256":"257d8f91a6628fd23b56a5d9f123261d0b29a47d2b7f7988232c6b77eb65714e","abstract_canon_sha256":"f3797ab2132e022ff329dbec6d8c7102e0362c74391f4cc6b10837c326361cf4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:36:15.687253Z","signature_b64":"qeWx9meU36nEIO0fhjvvGmg32Y3hTHyWx3Y/9zuq0Ol8fjqnnPvK7b7CQnOtIpmKgEpKvkaLD36uHF7/gqyvCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c662180c9c82d3403720529e580ee9fd1abf5571130acbc082094fe152aa6bd8","last_reissued_at":"2026-07-05T09:36:15.686721Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:36:15.686721Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ImOV3D: Learning Open-Vocabulary Point Clouds 3D Object Detection from Only 2D Images","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Li Yi, Timing Yang, Yuanliang Ju","submitted_at":"2024-10-31T15:02:05Z","abstract_excerpt":"Open-vocabulary 3D object detection (OV-3Det) aims to generalize beyond the limited number of base categories labeled during the training phase. The biggest bottleneck is the scarcity of annotated 3D data, whereas 2D image datasets are abundant and richly annotated. Consequently, it is intuitive to leverage the wealth of annotations in 2D images to alleviate the inherent data scarcity in OV-3Det. In this paper, we push the task setup to its limits by exploring the potential of using solely 2D images to learn OV-3Det. The major challenges for this setup is the modality gap between training imag"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.24001","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.24001/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.24001","created_at":"2026-07-05T09:36:15.686778+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.24001v1","created_at":"2026-07-05T09:36:15.686778+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.24001","created_at":"2026-07-05T09:36:15.686778+00:00"},{"alias_kind":"pith_short_12","alias_value":"YZRBQDE4QLJU","created_at":"2026-07-05T09:36:15.686778+00:00"},{"alias_kind":"pith_short_16","alias_value":"YZRBQDE4QLJUANZA","created_at":"2026-07-05T09:36:15.686778+00:00"},{"alias_kind":"pith_short_8","alias_value":"YZRBQDE4","created_at":"2026-07-05T09:36:15.686778+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.19492","citing_title":"Diorama: Unleashing Zero-shot Single-view 3D Indoor Scene Modeling","ref_index":86,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YZRBQDE4QLJUANZAKKPFQDXJ7U","json":"https://pith.science/pith/YZRBQDE4QLJUANZAKKPFQDXJ7U.json","graph_json":"https://pith.science/api/pith-number/YZRBQDE4QLJUANZAKKPFQDXJ7U/graph.json","events_json":"https://pith.science/api/pith-number/YZRBQDE4QLJUANZAKKPFQDXJ7U/events.json","paper":"https://pith.science/paper/YZRBQDE4"},"agent_actions":{"view_html":"https://pith.science/pith/YZRBQDE4QLJUANZAKKPFQDXJ7U","download_json":"https://pith.science/pith/YZRBQDE4QLJUANZAKKPFQDXJ7U.json","view_paper":"https://pith.science/paper/YZRBQDE4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.24001&json=true","fetch_graph":"https://pith.science/api/pith-number/YZRBQDE4QLJUANZAKKPFQDXJ7U/graph.json","fetch_events":"https://pith.science/api/pith-number/YZRBQDE4QLJUANZAKKPFQDXJ7U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YZRBQDE4QLJUANZAKKPFQDXJ7U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YZRBQDE4QLJUANZAKKPFQDXJ7U/action/storage_attestation","attest_author":"https://pith.science/pith/YZRBQDE4QLJUANZAKKPFQDXJ7U/action/author_attestation","sign_citation":"https://pith.science/pith/YZRBQDE4QLJUANZAKKPFQDXJ7U/action/citation_signature","submit_replication":"https://pith.science/pith/YZRBQDE4QLJUANZAKKPFQDXJ7U/action/replication_record"}},"created_at":"2026-07-05T09:36:15.686778+00:00","updated_at":"2026-07-05T09:36:15.686778+00:00"}