{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:KLY7P5BYMTM46F57IUR7UGZPYW","short_pith_number":"pith:KLY7P5BY","schema_version":"1.0","canonical_sha256":"52f1f7f43864d9cf17bf4523fa1b2fc59bc0cf4908cee9f21b6ccfcde94d231f","source":{"kind":"arxiv","id":"2211.08332","version":4},"attestation_state":"computed","paper":{"title":"Versatile Diffusion: Text, Images and Variations All in One Diffusion Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Eric Zhang, Humphrey Shi, Kai Wang, Xingqian Xu, Zhangyang Wang","submitted_at":"2022-11-15T17:44:05Z","abstract_excerpt":"Recent advances in diffusion models have set an impressive milestone in many generation tasks, and trending works such as DALL-E2, Imagen, and Stable Diffusion have attracted great interest. Despite the rapid landscape changes, recent new approaches focus on extensions and performance rather than capacity, thus requiring separate models for separate tasks. In this work, we expand the existing single-flow diffusion pipeline into a multi-task multimodal network, dubbed Versatile Diffusion (VD), that handles multiple flows of text-to-image, image-to-text, and variations in one unified model. The "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.08332","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-11-15T17:44:05Z","cross_cats_sorted":[],"title_canon_sha256":"74a12e02be8861e76fbf49a03ac499874500d209347b5abff66276f811d08b17","abstract_canon_sha256":"35a031105d5ea2d786ee9f86e2c7ca0e2366cf8400aad05b57e1ede2ee4c3b5c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:32:41.570159Z","signature_b64":"XBT9akdONQ1ZfUv/ToqZvpTcmQYlFkIYTgz8OABvR3tZO3Pde07ugiKLwl75n+iowiRVvbdoZv8dCPftNSQUBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"52f1f7f43864d9cf17bf4523fa1b2fc59bc0cf4908cee9f21b6ccfcde94d231f","last_reissued_at":"2026-07-05T07:32:41.569655Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:32:41.569655Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Versatile Diffusion: Text, Images and Variations All in One Diffusion Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Eric Zhang, Humphrey Shi, Kai Wang, Xingqian Xu, Zhangyang Wang","submitted_at":"2022-11-15T17:44:05Z","abstract_excerpt":"Recent advances in diffusion models have set an impressive milestone in many generation tasks, and trending works such as DALL-E2, Imagen, and Stable Diffusion have attracted great interest. Despite the rapid landscape changes, recent new approaches focus on extensions and performance rather than capacity, thus requiring separate models for separate tasks. In this work, we expand the existing single-flow diffusion pipeline into a multi-task multimodal network, dubbed Versatile Diffusion (VD), that handles multiple flows of text-to-image, image-to-text, and variations in one unified model. The "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.08332","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.08332/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.08332","created_at":"2026-07-05T07:32:41.569715+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.08332v4","created_at":"2026-07-05T07:32:41.569715+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.08332","created_at":"2026-07-05T07:32:41.569715+00:00"},{"alias_kind":"pith_short_12","alias_value":"KLY7P5BYMTM4","created_at":"2026-07-05T07:32:41.569715+00:00"},{"alias_kind":"pith_short_16","alias_value":"KLY7P5BYMTM46F57","created_at":"2026-07-05T07:32:41.569715+00:00"},{"alias_kind":"pith_short_8","alias_value":"KLY7P5BY","created_at":"2026-07-05T07:32:41.569715+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2306.04321","citing_title":"Generative Semantic Communication: Diffusion Models Beyond Bit Recovery","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20316","citing_title":"FullFlow: Upgrading Text-to-Image Flow Matching Models for Bidirectional Vision--Language Generation","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09622","citing_title":"Any2Any 3D Diffusion Models with Knowledge Transfer: A Radiotherapy Planning Study","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2306.09341","citing_title":"Human Preference Score v2: A Solid Benchmark for Evaluating Human Preferences of Text-to-Image Synthesis","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2308.06721","citing_title":"IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KLY7P5BYMTM46F57IUR7UGZPYW","json":"https://pith.science/pith/KLY7P5BYMTM46F57IUR7UGZPYW.json","graph_json":"https://pith.science/api/pith-number/KLY7P5BYMTM46F57IUR7UGZPYW/graph.json","events_json":"https://pith.science/api/pith-number/KLY7P5BYMTM46F57IUR7UGZPYW/events.json","paper":"https://pith.science/paper/KLY7P5BY"},"agent_actions":{"view_html":"https://pith.science/pith/KLY7P5BYMTM46F57IUR7UGZPYW","download_json":"https://pith.science/pith/KLY7P5BYMTM46F57IUR7UGZPYW.json","view_paper":"https://pith.science/paper/KLY7P5BY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.08332&json=true","fetch_graph":"https://pith.science/api/pith-number/KLY7P5BYMTM46F57IUR7UGZPYW/graph.json","fetch_events":"https://pith.science/api/pith-number/KLY7P5BYMTM46F57IUR7UGZPYW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KLY7P5BYMTM46F57IUR7UGZPYW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KLY7P5BYMTM46F57IUR7UGZPYW/action/storage_attestation","attest_author":"https://pith.science/pith/KLY7P5BYMTM46F57IUR7UGZPYW/action/author_attestation","sign_citation":"https://pith.science/pith/KLY7P5BYMTM46F57IUR7UGZPYW/action/citation_signature","submit_replication":"https://pith.science/pith/KLY7P5BYMTM46F57IUR7UGZPYW/action/replication_record"}},"created_at":"2026-07-05T07:32:41.569715+00:00","updated_at":"2026-07-05T07:32:41.569715+00:00"}