{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FS3IITZ53Z4FNZ4YENUY2UCXT2","short_pith_number":"pith:FS3IITZ5","schema_version":"1.0","canonical_sha256":"2cb6844f3dde7856e79823698d50579e87775dc753720133485c2febe1fed312","source":{"kind":"arxiv","id":"2312.02201","version":1},"attestation_state":"computed","paper":{"title":"ImageDream: Image-Prompt Multi-view Diffusion for 3D Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Peng Wang, Yichun Shi","submitted_at":"2023-12-02T20:41:27Z","abstract_excerpt":"We introduce \"ImageDream,\" an innovative image-prompt, multi-view diffusion model for 3D object generation. ImageDream stands out for its ability to produce 3D models of higher quality compared to existing state-of-the-art, image-conditioned methods. Our approach utilizes a canonical camera coordination for the objects in images, improving visual geometry accuracy. The model is designed with various levels of control at each block inside the diffusion model based on the input image, where global control shapes the overall object layout and local control fine-tunes the image details. The effect"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.02201","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-12-02T20:41:27Z","cross_cats_sorted":[],"title_canon_sha256":"947e09a84ca0add743d22f669579ac0c3297e291ecf159c9639204d55ec8dff0","abstract_canon_sha256":"52e4fbe5dd64bb32882ce99f79048f68e5fe488be0354ef78a8902f9659416a8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:20:14.294741Z","signature_b64":"PrjLIHmtar4KgyGHIJmJA6H9GJCwkoJdHYZFwH32BKQHtEQ0vELbZXWy2A+YEXNSvfwR6I2wtIfzsYdgTjzHAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2cb6844f3dde7856e79823698d50579e87775dc753720133485c2febe1fed312","last_reissued_at":"2026-07-05T07:20:14.294285Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:20:14.294285Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ImageDream: Image-Prompt Multi-view Diffusion for 3D Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Peng Wang, Yichun Shi","submitted_at":"2023-12-02T20:41:27Z","abstract_excerpt":"We introduce \"ImageDream,\" an innovative image-prompt, multi-view diffusion model for 3D object generation. ImageDream stands out for its ability to produce 3D models of higher quality compared to existing state-of-the-art, image-conditioned methods. Our approach utilizes a canonical camera coordination for the objects in images, improving visual geometry accuracy. The model is designed with various levels of control at each block inside the diffusion model based on the input image, where global control shapes the overall object layout and local control fine-tunes the image details. The effect"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.02201","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.02201/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.02201","created_at":"2026-07-05T07:20:14.294347+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.02201v1","created_at":"2026-07-05T07:20:14.294347+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.02201","created_at":"2026-07-05T07:20:14.294347+00:00"},{"alias_kind":"pith_short_12","alias_value":"FS3IITZ53Z4F","created_at":"2026-07-05T07:20:14.294347+00:00"},{"alias_kind":"pith_short_16","alias_value":"FS3IITZ53Z4FNZ4Y","created_at":"2026-07-05T07:20:14.294347+00:00"},{"alias_kind":"pith_short_8","alias_value":"FS3IITZ5","created_at":"2026-07-05T07:20:14.294347+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":27,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25300","citing_title":"HiFiVe: High-Fidelity Vehicle Generation Leveraging Auto-Regressive 2D Generative Priors","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24876","citing_title":"FLAT: Feedforward Latent Triangle Splatting for Geometrically Accurate Scene Generation","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24257","citing_title":"3DCarGen: Scalable 3D Car Generation via 3D-consistent Multi-view Synthesis","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21596","citing_title":"$\\phi$-Scene: Physically Grounded Image-to-3D Scene Reconstruction","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17030","citing_title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","ref_index":287,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02516","citing_title":"Alignment Is All You Need For X-to-4D Generation","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24257","citing_title":"3DCarGen: Scalable 3D Car Generation via 3D-consistent Multi-view Synthesis","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25300","citing_title":"HiFiVe: High-Fidelity Vehicle Generation Leveraging Auto-Regressive 2D Generative Priors","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29964","citing_title":"Variance Reduction on the Camera Axis: Multi-View Score Distillation for 3D","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2501.12202","citing_title":"Hunyuan3D 2.0: Scaling Diffusion Models for High Resolution Textured 3D Assets Generation","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18010","citing_title":"Functionalization via Structure Completion and Motion Rectification","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2405.10314","citing_title":"CAT3D: Create Anything in 3D with Multi-View Diffusion Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16807","citing_title":"DecoRec: Decomposed 3D Scene Reconstruction from Single-View Images via Object-Level Diffusion","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16990","citing_title":"DreamEdit3D: Personalization of Multi-View Diffusion Models for 3D Editing","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2509.07435","citing_title":"DreamLifting: A Plug-in Module Lifting MV Diffusion Models for 3D Asset Generation","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2502.06608","citing_title":"TripoSG: High-Fidelity 3D Shape Synthesis using Large-Scale Rectified Flow Models","ref_index":188,"is_internal_anchor":false},{"citing_arxiv_id":"2506.16504","citing_title":"Hunyuan3D 2.5: Towards High-Fidelity 3D Assets Generation with Ultimate Details","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13838","citing_title":"R-DMesh: Video-Guided 3D Animation via Rectified Dynamic Mesh Flow","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13838","citing_title":"R-DMesh: Video-Guided 3D Animation via Rectified Dynamic Mesh Flow","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2404.07191","citing_title":"InstantMesh: Efficient 3D Mesh Generation from a Single Image with Sparse-view Large Reconstruction Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27504","citing_title":"REVIVE 3D: Refinement via Encoded Voluminous Inflated prior for Volume Enhancement","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26917","citing_title":"AnimateAnyMesh++: A Flexible 4D Foundation Model for High-Fidelity Text-Driven Mesh Animation","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04527","citing_title":"Velox: Learning Representations of 4D Geometry and Appearance","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00345","citing_title":"Pose-Aware Diffusion for 3D Generation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18468","citing_title":"Asset Harvester: Extracting 3D Assets from Autonomous Driving Logs for Simulation","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FS3IITZ53Z4FNZ4YENUY2UCXT2","json":"https://pith.science/pith/FS3IITZ53Z4FNZ4YENUY2UCXT2.json","graph_json":"https://pith.science/api/pith-number/FS3IITZ53Z4FNZ4YENUY2UCXT2/graph.json","events_json":"https://pith.science/api/pith-number/FS3IITZ53Z4FNZ4YENUY2UCXT2/events.json","paper":"https://pith.science/paper/FS3IITZ5"},"agent_actions":{"view_html":"https://pith.science/pith/FS3IITZ53Z4FNZ4YENUY2UCXT2","download_json":"https://pith.science/pith/FS3IITZ53Z4FNZ4YENUY2UCXT2.json","view_paper":"https://pith.science/paper/FS3IITZ5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.02201&json=true","fetch_graph":"https://pith.science/api/pith-number/FS3IITZ53Z4FNZ4YENUY2UCXT2/graph.json","fetch_events":"https://pith.science/api/pith-number/FS3IITZ53Z4FNZ4YENUY2UCXT2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FS3IITZ53Z4FNZ4YENUY2UCXT2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FS3IITZ53Z4FNZ4YENUY2UCXT2/action/storage_attestation","attest_author":"https://pith.science/pith/FS3IITZ53Z4FNZ4YENUY2UCXT2/action/author_attestation","sign_citation":"https://pith.science/pith/FS3IITZ53Z4FNZ4YENUY2UCXT2/action/citation_signature","submit_replication":"https://pith.science/pith/FS3IITZ53Z4FNZ4YENUY2UCXT2/action/replication_record"}},"created_at":"2026-07-05T07:20:14.294347+00:00","updated_at":"2026-07-05T07:20:14.294347+00:00"}