{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3QAD2RBFB6SKEJBTJOPYOKXT7R","short_pith_number":"pith:3QAD2RBF","schema_version":"1.0","canonical_sha256":"dc003d44250fa4a224334b9f872af3fc6e52fda67efa498256d17b5554234496","source":{"kind":"arxiv","id":"2405.14832","version":2},"attestation_state":"computed","paper":{"title":"Direct3D: Scalable Image-to-3D Generation via 3D Latent Diffusion Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feihu Zhang, Jingxi Xu, Philip Torr, Shuang Wu, Xun Cao, Yao Yao, Yifei Zeng, Youtian Lin","submitted_at":"2024-05-23T17:49:37Z","abstract_excerpt":"Generating high-quality 3D assets from text and images has long been challenging, primarily due to the absence of scalable 3D representations capable of capturing intricate geometry distributions. In this work, we introduce Direct3D, a native 3D generative model scalable to in-the-wild input images, without requiring a multiview diffusion model or SDS optimization. Our approach comprises two primary components: a Direct 3D Variational Auto-Encoder (D3D-VAE) and a Direct 3D Diffusion Transformer (D3D-DiT). D3D-VAE efficiently encodes high-resolution 3D shapes into a compact and continuous laten"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14832","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-23T17:49:37Z","cross_cats_sorted":[],"title_canon_sha256":"90f4a490782181e24e2d8f8857d377bcf0ebf0c7d96fe7a164901aa64c305117","abstract_canon_sha256":"8557eec00cab809be386bd992daa965fecf1f6aa2d6837b2b025a302b3be3ff3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:54.620905Z","signature_b64":"Ts43Oi1oZl5S79xejGVsbsIy4MbfLNGtLpbUpp3M1W/yHqYRi8UfHP3RXkD5oiV3VeIW9ZHxwagpk5pptg4uDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dc003d44250fa4a224334b9f872af3fc6e52fda67efa498256d17b5554234496","last_reissued_at":"2026-07-05T08:25:54.620369Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:54.620369Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Direct3D: Scalable Image-to-3D Generation via 3D Latent Diffusion Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feihu Zhang, Jingxi Xu, Philip Torr, Shuang Wu, Xun Cao, Yao Yao, Yifei Zeng, Youtian Lin","submitted_at":"2024-05-23T17:49:37Z","abstract_excerpt":"Generating high-quality 3D assets from text and images has long been challenging, primarily due to the absence of scalable 3D representations capable of capturing intricate geometry distributions. In this work, we introduce Direct3D, a native 3D generative model scalable to in-the-wild input images, without requiring a multiview diffusion model or SDS optimization. Our approach comprises two primary components: a Direct 3D Variational Auto-Encoder (D3D-VAE) and a Direct 3D Diffusion Transformer (D3D-DiT). D3D-VAE efficiently encodes high-resolution 3D shapes into a compact and continuous laten"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14832","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14832/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14832","created_at":"2026-07-05T08:25:54.620431+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14832v2","created_at":"2026-07-05T08:25:54.620431+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14832","created_at":"2026-07-05T08:25:54.620431+00:00"},{"alias_kind":"pith_short_12","alias_value":"3QAD2RBFB6SK","created_at":"2026-07-05T08:25:54.620431+00:00"},{"alias_kind":"pith_short_16","alias_value":"3QAD2RBFB6SKEJBT","created_at":"2026-07-05T08:25:54.620431+00:00"},{"alias_kind":"pith_short_8","alias_value":"3QAD2RBF","created_at":"2026-07-05T08:25:54.620431+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07117","citing_title":"Native3D: End-to-End 3D Scene Generation via Unified Mesh-Texture Modeling and Semantic Alignment","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00522","citing_title":"Restore3D: Breathing Life into Broken Objects with Shape and Texture Restoration","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2606.32036","citing_title":"PointSplat: Compact Gaussian Splatting via Human-Centric Prediction","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28060","citing_title":"ReScene: Structured Indoor Scene Reconstruction from Multi-View Captures","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2501.12202","citing_title":"Hunyuan3D 2.0: Scaling Diffusion Models for High Resolution Textured 3D Assets Generation","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2512.14692","citing_title":"Native and Compact Structured Latents for 3D Generation","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21121","citing_title":"ROAR-3D: Routing Arbitrary Views for High-Fidelity 3D Generation","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2502.06608","citing_title":"TripoSG: High-Fidelity 3D Shape Synthesis using Large-Scale Rectified Flow Models","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2412.01506","citing_title":"Structured 3D Latents for Scalable and Versatile 3D Generation","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2603.11633","citing_title":"MV-SAM3D: Adaptive Multi-View Fusion for Layout-Aware 3D Generation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2603.16869","citing_title":"SegviGen: Repurposing 3D Generative Model for Part Segmentation","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04527","citing_title":"Velox: Learning Representations of 4D Geometry and Appearance","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18468","citing_title":"Asset Harvester: Extracting 3D Assets from Autonomous Driving Logs for Simulation","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10789","citing_title":"ReplicateAnyScene: Zero-Shot Video-to-3D Composition via Textual-Visual-Spatial Alignment","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3QAD2RBFB6SKEJBTJOPYOKXT7R","json":"https://pith.science/pith/3QAD2RBFB6SKEJBTJOPYOKXT7R.json","graph_json":"https://pith.science/api/pith-number/3QAD2RBFB6SKEJBTJOPYOKXT7R/graph.json","events_json":"https://pith.science/api/pith-number/3QAD2RBFB6SKEJBTJOPYOKXT7R/events.json","paper":"https://pith.science/paper/3QAD2RBF"},"agent_actions":{"view_html":"https://pith.science/pith/3QAD2RBFB6SKEJBTJOPYOKXT7R","download_json":"https://pith.science/pith/3QAD2RBFB6SKEJBTJOPYOKXT7R.json","view_paper":"https://pith.science/paper/3QAD2RBF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14832&json=true","fetch_graph":"https://pith.science/api/pith-number/3QAD2RBFB6SKEJBTJOPYOKXT7R/graph.json","fetch_events":"https://pith.science/api/pith-number/3QAD2RBFB6SKEJBTJOPYOKXT7R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3QAD2RBFB6SKEJBTJOPYOKXT7R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3QAD2RBFB6SKEJBTJOPYOKXT7R/action/storage_attestation","attest_author":"https://pith.science/pith/3QAD2RBFB6SKEJBTJOPYOKXT7R/action/author_attestation","sign_citation":"https://pith.science/pith/3QAD2RBFB6SKEJBTJOPYOKXT7R/action/citation_signature","submit_replication":"https://pith.science/pith/3QAD2RBFB6SKEJBTJOPYOKXT7R/action/replication_record"}},"created_at":"2026-07-05T08:25:54.620431+00:00","updated_at":"2026-07-05T08:25:54.620431+00:00"}