{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3YDXJGXX4QZ46PDNTMNVRXGYAN","short_pith_number":"pith:3YDXJGXX","schema_version":"1.0","canonical_sha256":"de07749af7e433cf3c6d9b1b58dcd8034f09f5d86fcdc6ada0807d1ae20564a3","source":{"kind":"arxiv","id":"2505.17412","version":2},"attestation_state":"computed","paper":{"title":"Direct3D-S2: Gigascale 3D Generation Made Easy with Spatial Sparse Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feihu Zhang, Jiachen Qian, Philip Torr, Shuang Wu, Siyu Zhu, Xun Cao, Yajie Bao, Yao Yao, Yifei Zeng, Yikang Yang, Youtian Lin","submitted_at":"2025-05-23T02:58:01Z","abstract_excerpt":"Generating high-resolution 3D shapes using volumetric representations such as Signed Distance Functions (SDFs) presents substantial computational and memory challenges. We introduce Direct3D-S2, a scalable 3D generation framework based on sparse volumes that achieves superior output quality with dramatically reduced training costs. Our key innovation is the Spatial Sparse Attention (SSA) mechanism, which greatly enhances the efficiency of Diffusion Transformer (DiT) computations on sparse volumetric data. SSA allows the model to effectively process large token sets within sparse volumes, subst"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.17412","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-23T02:58:01Z","cross_cats_sorted":[],"title_canon_sha256":"40e710b9c950b8f89f4577759c88d4f1627d70b8038e77ec48b98102c2693c6d","abstract_canon_sha256":"7be06c2bc2010b4808f2478db52990c24ff209169f47116ebb32fec97e103728"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:47.823692Z","signature_b64":"1c7hu6g+zjAM7TMpKAGeeRGap+YMGrMkOvYLPSlv2GruPhsYaVt+iKhbp5AivB14aM6u7cxju9dFTxi6XI8yCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"de07749af7e433cf3c6d9b1b58dcd8034f09f5d86fcdc6ada0807d1ae20564a3","last_reissued_at":"2026-07-05T11:09:47.823184Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:47.823184Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Direct3D-S2: Gigascale 3D Generation Made Easy with Spatial Sparse Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feihu Zhang, Jiachen Qian, Philip Torr, Shuang Wu, Siyu Zhu, Xun Cao, Yajie Bao, Yao Yao, Yifei Zeng, Yikang Yang, Youtian Lin","submitted_at":"2025-05-23T02:58:01Z","abstract_excerpt":"Generating high-resolution 3D shapes using volumetric representations such as Signed Distance Functions (SDFs) presents substantial computational and memory challenges. We introduce Direct3D-S2, a scalable 3D generation framework based on sparse volumes that achieves superior output quality with dramatically reduced training costs. Our key innovation is the Spatial Sparse Attention (SSA) mechanism, which greatly enhances the efficiency of Diffusion Transformer (DiT) computations on sparse volumetric data. SSA allows the model to effectively process large token sets within sparse volumes, subst"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.17412","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.17412/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.17412","created_at":"2026-07-05T11:09:47.823242+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.17412v2","created_at":"2026-07-05T11:09:47.823242+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.17412","created_at":"2026-07-05T11:09:47.823242+00:00"},{"alias_kind":"pith_short_12","alias_value":"3YDXJGXX4QZ4","created_at":"2026-07-05T11:09:47.823242+00:00"},{"alias_kind":"pith_short_16","alias_value":"3YDXJGXX4QZ46PDN","created_at":"2026-07-05T11:09:47.823242+00:00"},{"alias_kind":"pith_short_8","alias_value":"3YDXJGXX","created_at":"2026-07-05T11:09:47.823242+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":29,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07187","citing_title":"EditVerse3D: High-Quality 3D Object Editing with Region-Aware Learning","ref_index":84,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24874","citing_title":"FLUX3D: High-Fidelity 3D Gaussian Generation with Diffusion-Aligned Sparse Representation","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24301","citing_title":"MM-TRELLIS: Point-Cloud Guided Multi-Modal 3D Vehicle Generation in Autonomous Driving","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20131","citing_title":"TriFlow: Generating Artist-Like 3D Mesh Topology via Nearest-Vertex Vector Fields","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12099","citing_title":"ISAP-3D: Identity-Slot Aligned Part-Aware 3D Generation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01222","citing_title":"Ink3D: Sculpting 3D Assets with Extremely Complex Textures via Video Generative Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04688","citing_title":"MeshWeaver: Sparse-Voxel-Guided Surface Weaving for Autoregressive Mesh Generation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07971","citing_title":"DVD: Discrete Voxel Diffusion for 3D Generation and Editing","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31680","citing_title":"ShellMaker: Language-Guided Exterior Completion under Structural Constraints","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26109","citing_title":"Helix4D: Complex 4D Mesh Generation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26182","citing_title":"BrickAnything: Geometry-Conditioned Buildable Brick Generation with Structure-Aware Tokenization","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29655","citing_title":"SuperVoxelGPT: Adaptive and Ordered 3D Tokenization for Autoregressive Shape Generation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00299","citing_title":"Real2SAM2Real: Generative 3D Caches as Complementary Context for Video Diffusion","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2512.14692","citing_title":"Native and Compact Structured Latents for 3D Generation","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15843","citing_title":"WorldAct: Activating Monolithic 3D Worlds into Interactive-Ready Object-Centric Scenes","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17853","citing_title":"CelloCut: Constructive Watertight Remeshing via Tetrahedral Cell Cuts","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2509.07435","citing_title":"DreamLifting: A Plug-in Module Lifting MV Diffusion Models for 3D Asset Generation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15442","citing_title":"Hunyuan3D 2.1: From Images to High-Fidelity 3D Assets with Production-Ready PBR Material","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01479","citing_title":"UniRecGen: Unifying Multi-View 3D Reconstruction and Generation","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03359","citing_title":"Mix3R: Mixing Feed-forward Reconstruction and Generative 3D Priors for Joint Multi-view Aligned 3D Reconstruction and Pose Estimation","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26917","citing_title":"AnimateAnyMesh++: A Flexible 4D Foundation Model for High-Fidelity Text-Driven Mesh Animation","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10922","citing_title":"Pixal3D: Pixel-Aligned 3D Generation from Images","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01382","citing_title":"Sparse Representation Learning for Vessels","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2511.16624","citing_title":"SAM 3D: 3Dfy Anything in Images","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07385","citing_title":"Velocity-Space 3D Asset Editing","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3YDXJGXX4QZ46PDNTMNVRXGYAN","json":"https://pith.science/pith/3YDXJGXX4QZ46PDNTMNVRXGYAN.json","graph_json":"https://pith.science/api/pith-number/3YDXJGXX4QZ46PDNTMNVRXGYAN/graph.json","events_json":"https://pith.science/api/pith-number/3YDXJGXX4QZ46PDNTMNVRXGYAN/events.json","paper":"https://pith.science/paper/3YDXJGXX"},"agent_actions":{"view_html":"https://pith.science/pith/3YDXJGXX4QZ46PDNTMNVRXGYAN","download_json":"https://pith.science/pith/3YDXJGXX4QZ46PDNTMNVRXGYAN.json","view_paper":"https://pith.science/paper/3YDXJGXX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.17412&json=true","fetch_graph":"https://pith.science/api/pith-number/3YDXJGXX4QZ46PDNTMNVRXGYAN/graph.json","fetch_events":"https://pith.science/api/pith-number/3YDXJGXX4QZ46PDNTMNVRXGYAN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3YDXJGXX4QZ46PDNTMNVRXGYAN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3YDXJGXX4QZ46PDNTMNVRXGYAN/action/storage_attestation","attest_author":"https://pith.science/pith/3YDXJGXX4QZ46PDNTMNVRXGYAN/action/author_attestation","sign_citation":"https://pith.science/pith/3YDXJGXX4QZ46PDNTMNVRXGYAN/action/citation_signature","submit_replication":"https://pith.science/pith/3YDXJGXX4QZ46PDNTMNVRXGYAN/action/replication_record"}},"created_at":"2026-07-05T11:09:47.823242+00:00","updated_at":"2026-07-05T11:09:47.823242+00:00"}