{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JASPZDM3JRKL3YKZH7TR2XRIXG","short_pith_number":"pith:JASPZDM3","schema_version":"1.0","canonical_sha256":"4824fc8d9b4c54bde1593fe71d5e28b9a316e694bf137314f67b77ce1a01e3ef","source":{"kind":"arxiv","id":"2411.16318","version":2},"attestation_state":"computed","paper":{"title":"One Diffusion to Generate Them All","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Aniruddha Kembhavi, Christopher Clark, Duong H. Le, Jiasen Lu, Ranjay Krishna, Sangho Lee, Stephan Mandt, Tuan Pham","submitted_at":"2024-11-25T12:11:05Z","abstract_excerpt":"We introduce OneDiffusion, a versatile, large-scale diffusion model that seamlessly supports bidirectional image synthesis and understanding across diverse tasks. It enables conditional generation from inputs such as text, depth, pose, layout, and semantic maps, while also handling tasks like image deblurring, upscaling, and reverse processes such as depth estimation and segmentation. Additionally, OneDiffusion allows for multi-view generation, camera pose estimation, and instant personalization using sequential image inputs. Our model takes a straightforward yet effective approach by treating"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.16318","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-25T12:11:05Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"87968e07ccc7624f9d565e25edcd8c233c7058622c17685edee6e157dc6eff81","abstract_canon_sha256":"30d9d1a55398a04f66cae443bbee9f6e9c4eb16a2ae9b3b5f80cb9c511886d16"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:48.257564Z","signature_b64":"T9DRLc0mxgi8RxI3OqOoo9bfK0s3v+4rFgI5/xs/TKBT8K0iE0JCv0qwhzLlfFKR3/dS+4grLuGGxwXtufAkCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4824fc8d9b4c54bde1593fe71d5e28b9a316e694bf137314f67b77ce1a01e3ef","last_reissued_at":"2026-07-05T11:20:48.257073Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:48.257073Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"One Diffusion to Generate Them All","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Aniruddha Kembhavi, Christopher Clark, Duong H. Le, Jiasen Lu, Ranjay Krishna, Sangho Lee, Stephan Mandt, Tuan Pham","submitted_at":"2024-11-25T12:11:05Z","abstract_excerpt":"We introduce OneDiffusion, a versatile, large-scale diffusion model that seamlessly supports bidirectional image synthesis and understanding across diverse tasks. It enables conditional generation from inputs such as text, depth, pose, layout, and semantic maps, while also handling tasks like image deblurring, upscaling, and reverse processes such as depth estimation and segmentation. Additionally, OneDiffusion allows for multi-view generation, camera pose estimation, and instant personalization using sequential image inputs. Our model takes a straightforward yet effective approach by treating"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.16318","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.16318/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.16318","created_at":"2026-07-05T11:20:48.257136+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.16318v2","created_at":"2026-07-05T11:20:48.257136+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.16318","created_at":"2026-07-05T11:20:48.257136+00:00"},{"alias_kind":"pith_short_12","alias_value":"JASPZDM3JRKL","created_at":"2026-07-05T11:20:48.257136+00:00"},{"alias_kind":"pith_short_16","alias_value":"JASPZDM3JRKL3YKZ","created_at":"2026-07-05T11:20:48.257136+00:00"},{"alias_kind":"pith_short_8","alias_value":"JASPZDM3","created_at":"2026-07-05T11:20:48.257136+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2508.14461","citing_title":"Ouroboros: Single-step Diffusion Models for Cycle-consistent Forward and Inverse Rendering","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08090","citing_title":"DSH-Bench: A Difficulty- and Scenario-Aware Benchmark with Hierarchical Subject Taxonomy for Subject-Driven Text-to-Image Generation","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00658","citing_title":"UniVidX: A Unified Multimodal Framework for Versatile Video Generation via Diffusion Priors","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JASPZDM3JRKL3YKZH7TR2XRIXG","json":"https://pith.science/pith/JASPZDM3JRKL3YKZH7TR2XRIXG.json","graph_json":"https://pith.science/api/pith-number/JASPZDM3JRKL3YKZH7TR2XRIXG/graph.json","events_json":"https://pith.science/api/pith-number/JASPZDM3JRKL3YKZH7TR2XRIXG/events.json","paper":"https://pith.science/paper/JASPZDM3"},"agent_actions":{"view_html":"https://pith.science/pith/JASPZDM3JRKL3YKZH7TR2XRIXG","download_json":"https://pith.science/pith/JASPZDM3JRKL3YKZH7TR2XRIXG.json","view_paper":"https://pith.science/paper/JASPZDM3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.16318&json=true","fetch_graph":"https://pith.science/api/pith-number/JASPZDM3JRKL3YKZH7TR2XRIXG/graph.json","fetch_events":"https://pith.science/api/pith-number/JASPZDM3JRKL3YKZH7TR2XRIXG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JASPZDM3JRKL3YKZH7TR2XRIXG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JASPZDM3JRKL3YKZH7TR2XRIXG/action/storage_attestation","attest_author":"https://pith.science/pith/JASPZDM3JRKL3YKZH7TR2XRIXG/action/author_attestation","sign_citation":"https://pith.science/pith/JASPZDM3JRKL3YKZH7TR2XRIXG/action/citation_signature","submit_replication":"https://pith.science/pith/JASPZDM3JRKL3YKZH7TR2XRIXG/action/replication_record"}},"created_at":"2026-07-05T11:20:48.257136+00:00","updated_at":"2026-07-05T11:20:48.257136+00:00"}