{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TNUDPXLY2Y6ES5NCPV7HYMLKDW","short_pith_number":"pith:TNUDPXLY","schema_version":"1.0","canonical_sha256":"9b6837dd78d63c4975a27d7e7c316a1dab5b66e3e6da2281c0f6b66058b7ff37","source":{"kind":"arxiv","id":"2312.02980","version":2},"attestation_state":"computed","paper":{"title":"GPT4Point: A Unified Framework for Point-Language Understanding and Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dahua Lin, Hengshuang Zhao, Jiaqi Wang, Tong Wu, Xiaoyang Wu, Ye Fang, Zeyi Sun, Zhangyang Qi","submitted_at":"2023-12-05T18:59:55Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have excelled in 2D image-text comprehension and image generation, but their understanding of the 3D world is notably deficient, limiting progress in 3D language understanding and generation. To solve this problem, we introduce GPT4Point, an innovative groundbreaking point-language multimodal model designed specifically for unified 3D object understanding and generation within the MLLM framework. GPT4Point as a powerful 3D MLLM seamlessly can execute a variety of point-text reference tasks such as point-cloud captioning and Q&A. Additionally, GPT4Point "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.02980","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-05T18:59:55Z","cross_cats_sorted":[],"title_canon_sha256":"d202fc1336d67273333570fffac04b85419f4ecc48daa4ca2a453eebd84d085e","abstract_canon_sha256":"980c16dc98162a6274d11a619d6e52de327fc9ee80db028089a139b6c4dcadfd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:16.246261Z","signature_b64":"v2QEC7IoRMOJuuL5KnuCFyELh3NP9OtUh1ZmGKvvPXahmtd2BNyrkSUHcwADE06AfMUEhDU36JtpHwLcnoKpBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9b6837dd78d63c4975a27d7e7c316a1dab5b66e3e6da2281c0f6b66058b7ff37","last_reissued_at":"2026-07-05T11:12:16.245854Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:16.245854Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GPT4Point: A Unified Framework for Point-Language Understanding and Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dahua Lin, Hengshuang Zhao, Jiaqi Wang, Tong Wu, Xiaoyang Wu, Ye Fang, Zeyi Sun, Zhangyang Qi","submitted_at":"2023-12-05T18:59:55Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have excelled in 2D image-text comprehension and image generation, but their understanding of the 3D world is notably deficient, limiting progress in 3D language understanding and generation. To solve this problem, we introduce GPT4Point, an innovative groundbreaking point-language multimodal model designed specifically for unified 3D object understanding and generation within the MLLM framework. GPT4Point as a powerful 3D MLLM seamlessly can execute a variety of point-text reference tasks such as point-cloud captioning and Q&A. Additionally, GPT4Point "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.02980","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.02980/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.02980","created_at":"2026-07-05T11:12:16.245914+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.02980v2","created_at":"2026-07-05T11:12:16.245914+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.02980","created_at":"2026-07-05T11:12:16.245914+00:00"},{"alias_kind":"pith_short_12","alias_value":"TNUDPXLY2Y6E","created_at":"2026-07-05T11:12:16.245914+00:00"},{"alias_kind":"pith_short_16","alias_value":"TNUDPXLY2Y6ES5NC","created_at":"2026-07-05T11:12:16.245914+00:00"},{"alias_kind":"pith_short_8","alias_value":"TNUDPXLY","created_at":"2026-07-05T11:12:16.245914+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.08503","citing_title":"Revisiting 3D LLM Benchmarks: Are We Really Testing 3D Capabilities?","ref_index":19,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TNUDPXLY2Y6ES5NCPV7HYMLKDW","json":"https://pith.science/pith/TNUDPXLY2Y6ES5NCPV7HYMLKDW.json","graph_json":"https://pith.science/api/pith-number/TNUDPXLY2Y6ES5NCPV7HYMLKDW/graph.json","events_json":"https://pith.science/api/pith-number/TNUDPXLY2Y6ES5NCPV7HYMLKDW/events.json","paper":"https://pith.science/paper/TNUDPXLY"},"agent_actions":{"view_html":"https://pith.science/pith/TNUDPXLY2Y6ES5NCPV7HYMLKDW","download_json":"https://pith.science/pith/TNUDPXLY2Y6ES5NCPV7HYMLKDW.json","view_paper":"https://pith.science/paper/TNUDPXLY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.02980&json=true","fetch_graph":"https://pith.science/api/pith-number/TNUDPXLY2Y6ES5NCPV7HYMLKDW/graph.json","fetch_events":"https://pith.science/api/pith-number/TNUDPXLY2Y6ES5NCPV7HYMLKDW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TNUDPXLY2Y6ES5NCPV7HYMLKDW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TNUDPXLY2Y6ES5NCPV7HYMLKDW/action/storage_attestation","attest_author":"https://pith.science/pith/TNUDPXLY2Y6ES5NCPV7HYMLKDW/action/author_attestation","sign_citation":"https://pith.science/pith/TNUDPXLY2Y6ES5NCPV7HYMLKDW/action/citation_signature","submit_replication":"https://pith.science/pith/TNUDPXLY2Y6ES5NCPV7HYMLKDW/action/replication_record"}},"created_at":"2026-07-05T11:12:16.245914+00:00","updated_at":"2026-07-05T11:12:16.245914+00:00"}