{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:M4ATGVBZIHNOKWK4S2UEI66OC3","short_pith_number":"pith:M4ATGVBZ","schema_version":"1.0","canonical_sha256":"670133543941dae5595c96a8447bce16cbbb4ece31ed49273e4b0486ca0cabc2","source":{"kind":"arxiv","id":"2411.09595","version":1},"attestation_state":"computed","paper":{"title":"LLaMA-Mesh: Unifying 3D Mesh Generation with Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Hang Su, Jonathan Lorraine, Jun Zhu, Sanja Fidler, Xiaohui Zeng, Yikai Wang, Zhengyi Wang","submitted_at":"2024-11-14T17:08:23Z","abstract_excerpt":"This work explores expanding the capabilities of large language models (LLMs) pretrained on text to generate 3D meshes within a unified model. This offers key advantages of (1) leveraging spatial knowledge already embedded in LLMs, derived from textual sources like 3D tutorials, and (2) enabling conversational 3D generation and mesh understanding. A primary challenge is effectively tokenizing 3D mesh data into discrete tokens that LLMs can process seamlessly. To address this, we introduce LLaMA-Mesh, a novel approach that represents the vertex coordinates and face definitions of 3D meshes as p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.09595","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-14T17:08:23Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV"],"title_canon_sha256":"1cd7a6591e4d5dc2f93511ea84e6150c0cfe2ab6e954e1530c5cd51d04a7f7af","abstract_canon_sha256":"e220f457b9d857e02461f9fdc5a5653ca6ab44810e51201550f80bf2a18ab34d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:35:29.254597Z","signature_b64":"+qrGXUb6bkPoVq5wWBIoK1zxeqwpmJr4bC2f7YbuOqxthYwzYnL2P4jpJe+yxT9PxRx1DS5jO7Cg1gVqFaEMAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"670133543941dae5595c96a8447bce16cbbb4ece31ed49273e4b0486ca0cabc2","last_reissued_at":"2026-07-05T09:35:29.254057Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:35:29.254057Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLaMA-Mesh: Unifying 3D Mesh Generation with Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Hang Su, Jonathan Lorraine, Jun Zhu, Sanja Fidler, Xiaohui Zeng, Yikai Wang, Zhengyi Wang","submitted_at":"2024-11-14T17:08:23Z","abstract_excerpt":"This work explores expanding the capabilities of large language models (LLMs) pretrained on text to generate 3D meshes within a unified model. This offers key advantages of (1) leveraging spatial knowledge already embedded in LLMs, derived from textual sources like 3D tutorials, and (2) enabling conversational 3D generation and mesh understanding. A primary challenge is effectively tokenizing 3D mesh data into discrete tokens that LLMs can process seamlessly. To address this, we introduce LLaMA-Mesh, a novel approach that represents the vertex coordinates and face definitions of 3D meshes as p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.09595","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.09595/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.09595","created_at":"2026-07-05T09:35:29.254146+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.09595v1","created_at":"2026-07-05T09:35:29.254146+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.09595","created_at":"2026-07-05T09:35:29.254146+00:00"},{"alias_kind":"pith_short_12","alias_value":"M4ATGVBZIHNO","created_at":"2026-07-05T09:35:29.254146+00:00"},{"alias_kind":"pith_short_16","alias_value":"M4ATGVBZIHNOKWK4","created_at":"2026-07-05T09:35:29.254146+00:00"},{"alias_kind":"pith_short_8","alias_value":"M4ATGVBZ","created_at":"2026-07-05T09:35:29.254146+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06565","citing_title":"ELSA3D: Elastic Semantic Anchoring for Unified 3D Understanding and Generation","ref_index":31,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24138","citing_title":"Sat2City v2: Native 3D City Asset Generation from a Single Satellite Image","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27305","citing_title":"Sculpting NeRF Geometry: Human-Preference Fine-Tuning of a 3D-Aware Face GAN","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31388","citing_title":"One Video, One World: Turning Monocular Video into Physical 4D Scenes","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21489","citing_title":"Variance Reduction for Expectations with Diffusion Teachers","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21572","citing_title":"PhysX-Omni: Unified Simulation-Ready Physical 3D Generation for Rigid, Deformable, and Articulated Objects","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18801","citing_title":"PartDiffuser: Part-wise 3D Mesh Generation via Discrete Diffusion","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21798","citing_title":"CG-MLLM: Captioning and Generating 3D content via Multi-modal Large Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27309","citing_title":"MeshTailor: Cutting Seams via Generative Mesh Traversal","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21489","citing_title":"Variance Reduction for Expectations with Diffusion Teachers","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17620","citing_title":"SynVA: A Modular Toolkit for Vessel Generation and Aneurysm Editing","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16745","citing_title":"EVA01: Unified Native 3D Understanding and Generation via Mixture-of-Transformers","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16813","citing_title":"QuadLink: Autoregressive Quad-Dominant Mesh Generation via Point-Relation Learning","ref_index":161,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14594","citing_title":"TOPOS: High-Fidelity and Efficient Industry-Grade 3D Head Generation","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03305","citing_title":"HVG-3D: Bridging Real and Simulation Domains for 3D-Conditional Hand-Object Interaction Video Synthesis","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01479","citing_title":"UniRecGen: Unifying Multi-View 3D Reconstruction and Generation","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10438","citing_title":"Beyond Spatial Compression: Interface-Centric Generative States for Open-World 3D Structure","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11792","citing_title":"LottieGPT: Tokenizing Vector Animation for Autoregressive Generation","ref_index":80,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M4ATGVBZIHNOKWK4S2UEI66OC3","json":"https://pith.science/pith/M4ATGVBZIHNOKWK4S2UEI66OC3.json","graph_json":"https://pith.science/api/pith-number/M4ATGVBZIHNOKWK4S2UEI66OC3/graph.json","events_json":"https://pith.science/api/pith-number/M4ATGVBZIHNOKWK4S2UEI66OC3/events.json","paper":"https://pith.science/paper/M4ATGVBZ"},"agent_actions":{"view_html":"https://pith.science/pith/M4ATGVBZIHNOKWK4S2UEI66OC3","download_json":"https://pith.science/pith/M4ATGVBZIHNOKWK4S2UEI66OC3.json","view_paper":"https://pith.science/paper/M4ATGVBZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.09595&json=true","fetch_graph":"https://pith.science/api/pith-number/M4ATGVBZIHNOKWK4S2UEI66OC3/graph.json","fetch_events":"https://pith.science/api/pith-number/M4ATGVBZIHNOKWK4S2UEI66OC3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M4ATGVBZIHNOKWK4S2UEI66OC3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M4ATGVBZIHNOKWK4S2UEI66OC3/action/storage_attestation","attest_author":"https://pith.science/pith/M4ATGVBZIHNOKWK4S2UEI66OC3/action/author_attestation","sign_citation":"https://pith.science/pith/M4ATGVBZIHNOKWK4S2UEI66OC3/action/citation_signature","submit_replication":"https://pith.science/pith/M4ATGVBZIHNOKWK4S2UEI66OC3/action/replication_record"}},"created_at":"2026-07-05T09:35:29.254146+00:00","updated_at":"2026-07-05T09:35:29.254146+00:00"}