{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:EBH6NUBVML4CFZRB7N5PZJTPSO","short_pith_number":"pith:EBH6NUBV","schema_version":"1.0","canonical_sha256":"204fe6d03562f822e621fb7afca66f938285b3b4dfdb67ad800e385f0e3e871d","source":{"kind":"arxiv","id":"2211.11682","version":2},"attestation_state":"computed","paper":{"title":"PointCLIP V2: Prompting CLIP and GPT for Powerful 3D Open-world Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bowei He, Peng Gao, Renrui Zhang, Shanghang Zhang, Xiangyang Zhu, Zipeng Qin, Ziyao Zeng, Ziyu Guo","submitted_at":"2022-11-21T17:52:43Z","abstract_excerpt":"Large-scale pre-trained models have shown promising open-world performance for both vision and language tasks. However, their transferred capacity on 3D point clouds is still limited and only constrained to the classification task. In this paper, we first collaborate CLIP and GPT to be a unified 3D open-world learner, named as PointCLIP V2, which fully unleashes their potential for zero-shot 3D classification, segmentation, and detection. To better align 3D data with the pre-trained language knowledge, PointCLIP V2 contains two key designs. For the visual end, we prompt CLIP via a shape projec"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.11682","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-11-21T17:52:43Z","cross_cats_sorted":[],"title_canon_sha256":"03a9c6e009f989d7ca3930d59b0217905ae1efdc0c7671b8b7beb84000d0088d","abstract_canon_sha256":"76074a457e1a881ab2c743e92b636c0bd4de53d5543f98f0926a1e6c646409e0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:44:47.354016Z","signature_b64":"GntwCXj+NUnmLCbYZSJe6vo9shx3LKDZePcp9FTGyABwpgi1+3PbGM3pp26pN8GDJHj2FqhwvJUvltc/pb7gDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"204fe6d03562f822e621fb7afca66f938285b3b4dfdb67ad800e385f0e3e871d","last_reissued_at":"2026-07-05T06:44:47.353523Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:44:47.353523Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PointCLIP V2: Prompting CLIP and GPT for Powerful 3D Open-world Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bowei He, Peng Gao, Renrui Zhang, Shanghang Zhang, Xiangyang Zhu, Zipeng Qin, Ziyao Zeng, Ziyu Guo","submitted_at":"2022-11-21T17:52:43Z","abstract_excerpt":"Large-scale pre-trained models have shown promising open-world performance for both vision and language tasks. However, their transferred capacity on 3D point clouds is still limited and only constrained to the classification task. In this paper, we first collaborate CLIP and GPT to be a unified 3D open-world learner, named as PointCLIP V2, which fully unleashes their potential for zero-shot 3D classification, segmentation, and detection. To better align 3D data with the pre-trained language knowledge, PointCLIP V2 contains two key designs. For the visual end, we prompt CLIP via a shape projec"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.11682","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.11682/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.11682","created_at":"2026-07-05T06:44:47.353582+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.11682v2","created_at":"2026-07-05T06:44:47.353582+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.11682","created_at":"2026-07-05T06:44:47.353582+00:00"},{"alias_kind":"pith_short_12","alias_value":"EBH6NUBVML4C","created_at":"2026-07-05T06:44:47.353582+00:00"},{"alias_kind":"pith_short_16","alias_value":"EBH6NUBVML4CFZRB","created_at":"2026-07-05T06:44:47.353582+00:00"},{"alias_kind":"pith_short_8","alias_value":"EBH6NUBV","created_at":"2026-07-05T06:44:47.353582+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":292,"is_internal_anchor":false},{"citing_arxiv_id":"2311.07575","citing_title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":200,"is_internal_anchor":false},{"citing_arxiv_id":"2304.15010","citing_title":"LLaMA-Adapter V2: Parameter-Efficient Visual Instruction Model","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2303.16199","citing_title":"LLaMA-Adapter: Efficient Fine-tuning of Language Models with Zero-init Attention","ref_index":192,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22658","citing_title":"PASR: Pose-Aware 3D Shape Retrieval from Occluded Single Views","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EBH6NUBVML4CFZRB7N5PZJTPSO","json":"https://pith.science/pith/EBH6NUBVML4CFZRB7N5PZJTPSO.json","graph_json":"https://pith.science/api/pith-number/EBH6NUBVML4CFZRB7N5PZJTPSO/graph.json","events_json":"https://pith.science/api/pith-number/EBH6NUBVML4CFZRB7N5PZJTPSO/events.json","paper":"https://pith.science/paper/EBH6NUBV"},"agent_actions":{"view_html":"https://pith.science/pith/EBH6NUBVML4CFZRB7N5PZJTPSO","download_json":"https://pith.science/pith/EBH6NUBVML4CFZRB7N5PZJTPSO.json","view_paper":"https://pith.science/paper/EBH6NUBV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.11682&json=true","fetch_graph":"https://pith.science/api/pith-number/EBH6NUBVML4CFZRB7N5PZJTPSO/graph.json","fetch_events":"https://pith.science/api/pith-number/EBH6NUBVML4CFZRB7N5PZJTPSO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EBH6NUBVML4CFZRB7N5PZJTPSO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EBH6NUBVML4CFZRB7N5PZJTPSO/action/storage_attestation","attest_author":"https://pith.science/pith/EBH6NUBVML4CFZRB7N5PZJTPSO/action/author_attestation","sign_citation":"https://pith.science/pith/EBH6NUBVML4CFZRB7N5PZJTPSO/action/citation_signature","submit_replication":"https://pith.science/pith/EBH6NUBVML4CFZRB7N5PZJTPSO/action/replication_record"}},"created_at":"2026-07-05T06:44:47.353582+00:00","updated_at":"2026-07-05T06:44:47.353582+00:00"}