{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:B7HT7HSLMISVWD7I2ARZWJOUNQ","short_pith_number":"pith:B7HT7HSL","schema_version":"1.0","canonical_sha256":"0fcf3f9e4b62255b0fe8d0239b25d46c3d51bd68d164684baed5e425a7c0c891","source":{"kind":"arxiv","id":"2406.11579","version":3},"attestation_state":"computed","paper":{"title":"Duoduo CLIP: Efficient 3D Understanding with Multi-View Images","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Angel X. Chang, Han-Hung Lee, Yiming Zhang","submitted_at":"2024-06-17T14:16:12Z","abstract_excerpt":"We introduce Duoduo CLIP, a model for 3D representation learning that learns shape encodings from multi-view images instead of point clouds. The choice of multi-view images allows us to leverage 2D priors from off-the-shelf CLIP models to facilitate fine-tuning with 3D data. Our approach not only shows better generalization compared to existing point cloud methods, but also reduces GPU requirements and training time. In addition, the model is modified with cross-view attention to leverage information across multiple frames of the object which further boosts performance. Notably, our model is p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.11579","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-17T14:16:12Z","cross_cats_sorted":[],"title_canon_sha256":"dfacb9c56848037e7d825da6acfcb7ea0f0323f0fb002a9bcd9408aaeffac7a9","abstract_canon_sha256":"58d8ea07157cea1685c77cadd6ccb37d37c4b54c878376eede091d6fe7003b48"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:35:33.712906Z","signature_b64":"pIfmCOXpDiQKyxm8Z1pX61mAFPsULQ39NhaMm9rbQomYLXRkzLZ/2kRuBmiJvGXLLqgo+vExeV6DOqAys8VlCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0fcf3f9e4b62255b0fe8d0239b25d46c3d51bd68d164684baed5e425a7c0c891","last_reissued_at":"2026-07-05T10:35:33.712371Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:35:33.712371Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Duoduo CLIP: Efficient 3D Understanding with Multi-View Images","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Angel X. Chang, Han-Hung Lee, Yiming Zhang","submitted_at":"2024-06-17T14:16:12Z","abstract_excerpt":"We introduce Duoduo CLIP, a model for 3D representation learning that learns shape encodings from multi-view images instead of point clouds. The choice of multi-view images allows us to leverage 2D priors from off-the-shelf CLIP models to facilitate fine-tuning with 3D data. Our approach not only shows better generalization compared to existing point cloud methods, but also reduces GPU requirements and training time. In addition, the model is modified with cross-view attention to leverage information across multiple frames of the object which further boosts performance. Notably, our model is p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.11579","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.11579/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.11579","created_at":"2026-07-05T10:35:33.712443+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.11579v3","created_at":"2026-07-05T10:35:33.712443+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.11579","created_at":"2026-07-05T10:35:33.712443+00:00"},{"alias_kind":"pith_short_12","alias_value":"B7HT7HSLMISV","created_at":"2026-07-05T10:35:33.712443+00:00"},{"alias_kind":"pith_short_16","alias_value":"B7HT7HSLMISVWD7I","created_at":"2026-07-05T10:35:33.712443+00:00"},{"alias_kind":"pith_short_8","alias_value":"B7HT7HSL","created_at":"2026-07-05T10:35:33.712443+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.02546","citing_title":"RGB-Pointmap Pretraining for Unified 3D Scene Understanding","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B7HT7HSLMISVWD7I2ARZWJOUNQ","json":"https://pith.science/pith/B7HT7HSLMISVWD7I2ARZWJOUNQ.json","graph_json":"https://pith.science/api/pith-number/B7HT7HSLMISVWD7I2ARZWJOUNQ/graph.json","events_json":"https://pith.science/api/pith-number/B7HT7HSLMISVWD7I2ARZWJOUNQ/events.json","paper":"https://pith.science/paper/B7HT7HSL"},"agent_actions":{"view_html":"https://pith.science/pith/B7HT7HSLMISVWD7I2ARZWJOUNQ","download_json":"https://pith.science/pith/B7HT7HSLMISVWD7I2ARZWJOUNQ.json","view_paper":"https://pith.science/paper/B7HT7HSL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.11579&json=true","fetch_graph":"https://pith.science/api/pith-number/B7HT7HSLMISVWD7I2ARZWJOUNQ/graph.json","fetch_events":"https://pith.science/api/pith-number/B7HT7HSLMISVWD7I2ARZWJOUNQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B7HT7HSLMISVWD7I2ARZWJOUNQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B7HT7HSLMISVWD7I2ARZWJOUNQ/action/storage_attestation","attest_author":"https://pith.science/pith/B7HT7HSLMISVWD7I2ARZWJOUNQ/action/author_attestation","sign_citation":"https://pith.science/pith/B7HT7HSLMISVWD7I2ARZWJOUNQ/action/citation_signature","submit_replication":"https://pith.science/pith/B7HT7HSLMISVWD7I2ARZWJOUNQ/action/replication_record"}},"created_at":"2026-07-05T10:35:33.712443+00:00","updated_at":"2026-07-05T10:35:33.712443+00:00"}