{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IGTKIYDJZ7CDIC57CG75GC5C67","short_pith_number":"pith:IGTKIYDJ","schema_version":"1.0","canonical_sha256":"41a6a46069cfc4340bbf11bfd30ba2f7cc96a6537b7579a8831d708aefa2ba06","source":{"kind":"arxiv","id":"2308.16911","version":3},"attestation_state":"computed","paper":{"title":"PointLLM: Empowering Large Language Models to Understand Point Clouds","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Dahua Lin, Jiangmiao Pang, Runsen Xu, Tai Wang, Xiaolong Wang, Yilun Chen","submitted_at":"2023-08-31T17:59:46Z","abstract_excerpt":"The unprecedented advancements in Large Language Models (LLMs) have shown a profound impact on natural language processing but are yet to fully embrace the realm of 3D understanding. This paper introduces PointLLM, a preliminary effort to fill this gap, enabling LLMs to understand point clouds and offering a new avenue beyond 2D visual data. PointLLM understands colored object point clouds with human instructions and generates contextually appropriate responses, illustrating its grasp of point clouds and common sense. Specifically, it leverages a point cloud encoder with a powerful LLM to effe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.16911","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2023-08-31T17:59:46Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"1ebc0b5c9a6118df7e55468351e66a035b265722315fac42cea21d7101d36dda","abstract_canon_sha256":"c19c69aa103ce0a4beac8a1fa9fc9339a9869e1817facf8d65cc0b3e4244df7d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:04:07.481871Z","signature_b64":"nez8CrKew6ZkM0PMVMFzkga2p2350hzCId15s8QYQk0tKxNJrBdgTi0pIs6BtdDQ9SsbgqWv/2RzWY7xH7BWBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"41a6a46069cfc4340bbf11bfd30ba2f7cc96a6537b7579a8831d708aefa2ba06","last_reissued_at":"2026-07-05T09:04:07.481402Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:04:07.481402Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PointLLM: Empowering Large Language Models to Understand Point Clouds","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Dahua Lin, Jiangmiao Pang, Runsen Xu, Tai Wang, Xiaolong Wang, Yilun Chen","submitted_at":"2023-08-31T17:59:46Z","abstract_excerpt":"The unprecedented advancements in Large Language Models (LLMs) have shown a profound impact on natural language processing but are yet to fully embrace the realm of 3D understanding. This paper introduces PointLLM, a preliminary effort to fill this gap, enabling LLMs to understand point clouds and offering a new avenue beyond 2D visual data. PointLLM understands colored object point clouds with human instructions and generates contextually appropriate responses, illustrating its grasp of point clouds and common sense. Specifically, it leverages a point cloud encoder with a powerful LLM to effe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.16911","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.16911/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.16911","created_at":"2026-07-05T09:04:07.481458+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.16911v3","created_at":"2026-07-05T09:04:07.481458+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.16911","created_at":"2026-07-05T09:04:07.481458+00:00"},{"alias_kind":"pith_short_12","alias_value":"IGTKIYDJZ7CD","created_at":"2026-07-05T09:04:07.481458+00:00"},{"alias_kind":"pith_short_16","alias_value":"IGTKIYDJZ7CDIC57","created_at":"2026-07-05T09:04:07.481458+00:00"},{"alias_kind":"pith_short_8","alias_value":"IGTKIYDJ","created_at":"2026-07-05T09:04:07.481458+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17888","citing_title":"MathVis-Fine: Aligning Visual Supervision with Necessity via Progressive Dependency-Guided Training for Multimodal Mathematical Reasoning","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04381","citing_title":"From Symbolic to Geometric: Enabling Spatial Reasoning in Large Language Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03460","citing_title":"From 3D Perception to Safety Reasoning: A Graph-Based Framework for Real-Time Underground Mine Monitoring","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21798","citing_title":"CG-MLLM: Captioning and Generating 3D content via Multi-modal Large Language Models","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2311.07575","citing_title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2403.14624","citing_title":"MathVerse: Does Your Multi-modal LLM Truly See the Diagrams in Visual Math Problems?","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":145,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27507","citing_title":"Chat-Scene++: Exploiting Context-Rich Object Identification for 3D LLM","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00663","citing_title":"Affordance Agent Harness: Verification-Gated Skill Orchestration","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2407.07895","citing_title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00663","citing_title":"Affordance Agent Harness: Verification-Gated Skill Orchestration","ref_index":76,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IGTKIYDJZ7CDIC57CG75GC5C67","json":"https://pith.science/pith/IGTKIYDJZ7CDIC57CG75GC5C67.json","graph_json":"https://pith.science/api/pith-number/IGTKIYDJZ7CDIC57CG75GC5C67/graph.json","events_json":"https://pith.science/api/pith-number/IGTKIYDJZ7CDIC57CG75GC5C67/events.json","paper":"https://pith.science/paper/IGTKIYDJ"},"agent_actions":{"view_html":"https://pith.science/pith/IGTKIYDJZ7CDIC57CG75GC5C67","download_json":"https://pith.science/pith/IGTKIYDJZ7CDIC57CG75GC5C67.json","view_paper":"https://pith.science/paper/IGTKIYDJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.16911&json=true","fetch_graph":"https://pith.science/api/pith-number/IGTKIYDJZ7CDIC57CG75GC5C67/graph.json","fetch_events":"https://pith.science/api/pith-number/IGTKIYDJZ7CDIC57CG75GC5C67/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IGTKIYDJZ7CDIC57CG75GC5C67/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IGTKIYDJZ7CDIC57CG75GC5C67/action/storage_attestation","attest_author":"https://pith.science/pith/IGTKIYDJZ7CDIC57CG75GC5C67/action/author_attestation","sign_citation":"https://pith.science/pith/IGTKIYDJZ7CDIC57CG75GC5C67/action/citation_signature","submit_replication":"https://pith.science/pith/IGTKIYDJZ7CDIC57CG75GC5C67/action/replication_record"}},"created_at":"2026-07-05T09:04:07.481458+00:00","updated_at":"2026-07-05T09:04:07.481458+00:00"}