{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5QW57EG2YQCUQJ6SZCXENLTSCM","short_pith_number":"pith:5QW57EG2","schema_version":"1.0","canonical_sha256":"ec2ddf90dac4054827d2c8ae46ae72133c9f09e60905640866368726538c6664","source":{"kind":"arxiv","id":"2503.20853","version":1},"attestation_state":"computed","paper":{"title":"Unified Multimodal Discrete Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Alexander Swerdlow, Deepak Pathak, Katerina Fragkiadaki, Mihir Prabhudesai, Siddharth Gandhi","submitted_at":"2025-03-26T17:59:51Z","abstract_excerpt":"Multimodal generative models that can understand and generate across multiple modalities are dominated by autoregressive (AR) approaches, which process tokens sequentially from left to right, or top to bottom. These models jointly handle images, text, video, and audio for various tasks such as image captioning, question answering, and image generation. In this work, we explore discrete diffusion models as a unified generative formulation in the joint text and image domain, building upon their recent success in text generation. Discrete diffusion models offer several advantages over AR models, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.20853","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-26T17:59:51Z","cross_cats_sorted":["cs.AI","cs.LG","cs.RO"],"title_canon_sha256":"4f165c1d6627727c68dd7ba23e5fa51154f37a6efb40b2b931ec204f634f57be","abstract_canon_sha256":"5162a78075c5f91a20df196412351d804824e6e2bbfc7e3bbd3a2b83dae73b6b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:39:49.202864Z","signature_b64":"0VdDvnpYvuRZu6gXHabKTzaOMuQAk8J9kncvRZQTh0IRiJ3wrPuDXyDBASSqcDimmSotwaCjNoqSFp169Y/vBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec2ddf90dac4054827d2c8ae46ae72133c9f09e60905640866368726538c6664","last_reissued_at":"2026-07-05T10:39:49.202394Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:39:49.202394Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unified Multimodal Discrete Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Alexander Swerdlow, Deepak Pathak, Katerina Fragkiadaki, Mihir Prabhudesai, Siddharth Gandhi","submitted_at":"2025-03-26T17:59:51Z","abstract_excerpt":"Multimodal generative models that can understand and generate across multiple modalities are dominated by autoregressive (AR) approaches, which process tokens sequentially from left to right, or top to bottom. These models jointly handle images, text, video, and audio for various tasks such as image captioning, question answering, and image generation. In this work, we explore discrete diffusion models as a unified generative formulation in the joint text and image domain, building upon their recent success in text generation. Discrete diffusion models offer several advantages over AR models, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.20853","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.20853/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.20853","created_at":"2026-07-05T10:39:49.202455+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.20853v1","created_at":"2026-07-05T10:39:49.202455+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.20853","created_at":"2026-07-05T10:39:49.202455+00:00"},{"alias_kind":"pith_short_12","alias_value":"5QW57EG2YQCU","created_at":"2026-07-05T10:39:49.202455+00:00"},{"alias_kind":"pith_short_16","alias_value":"5QW57EG2YQCUQJ6S","created_at":"2026-07-05T10:39:49.202455+00:00"},{"alias_kind":"pith_short_8","alias_value":"5QW57EG2","created_at":"2026-07-05T10:39:49.202455+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24333","citing_title":"UniTranslator: A Unified Multi-modal Framework for End-to-end In-Image Machine Translation","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07895","citing_title":"TBD-VLA: Temporal Block Diffusion Vision Language Action Model","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14531","citing_title":"Language Generation as Optimal Control: Closed-Loop Diffusion in Latent Control Space","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07971","citing_title":"DVD: Discrete Voxel Diffusion for 3D Generation and Editing","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31326","citing_title":"Bridging Video Understanding and Generation in a Unified Framework","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07079","citing_title":"AsyncPatch Diffusion: spatially-flexible image generation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21272","citing_title":"MONET: A Massive, Open, Non-redundant and Enriched Text-to-image dataset","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14531","citing_title":"Language Generation as Optimal Control: Closed-Loop Diffusion in Latent Control Space","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23606","citing_title":"Muddit: Liberating Generation Beyond Text-to-Image with a Unified Discrete Diffusion Model","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21912","citing_title":"Discrete Guidance Matching: Exact Guidance for Discrete Flow Matching","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2511.08416","citing_title":"Generative AI Meets 6G and Beyond: Diffusion Models for Semantic Communications","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14148","citing_title":"AsyncVLA: Asynchronous Flow Matching for Vision-Language-Action Models","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16933","citing_title":"LLaDA-V: Large Language Diffusion Models with Visual Instruction Tuning","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14531","citing_title":"Language Generation as Optimal Control: Closed-Loop Diffusion in Latent Control Space","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15564","citing_title":"Show-o2: Improved Native Unified Multimodal Models","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09291","citing_title":"dFlowGRPO: Rate-Aware Policy Optimization for Discrete Flow Models","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07971","citing_title":"DVD: Discrete Voxel Diffusion for 3D Generation and Editing","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5QW57EG2YQCUQJ6SZCXENLTSCM","json":"https://pith.science/pith/5QW57EG2YQCUQJ6SZCXENLTSCM.json","graph_json":"https://pith.science/api/pith-number/5QW57EG2YQCUQJ6SZCXENLTSCM/graph.json","events_json":"https://pith.science/api/pith-number/5QW57EG2YQCUQJ6SZCXENLTSCM/events.json","paper":"https://pith.science/paper/5QW57EG2"},"agent_actions":{"view_html":"https://pith.science/pith/5QW57EG2YQCUQJ6SZCXENLTSCM","download_json":"https://pith.science/pith/5QW57EG2YQCUQJ6SZCXENLTSCM.json","view_paper":"https://pith.science/paper/5QW57EG2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.20853&json=true","fetch_graph":"https://pith.science/api/pith-number/5QW57EG2YQCUQJ6SZCXENLTSCM/graph.json","fetch_events":"https://pith.science/api/pith-number/5QW57EG2YQCUQJ6SZCXENLTSCM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5QW57EG2YQCUQJ6SZCXENLTSCM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5QW57EG2YQCUQJ6SZCXENLTSCM/action/storage_attestation","attest_author":"https://pith.science/pith/5QW57EG2YQCUQJ6SZCXENLTSCM/action/author_attestation","sign_citation":"https://pith.science/pith/5QW57EG2YQCUQJ6SZCXENLTSCM/action/citation_signature","submit_replication":"https://pith.science/pith/5QW57EG2YQCUQJ6SZCXENLTSCM/action/replication_record"}},"created_at":"2026-07-05T10:39:49.202455+00:00","updated_at":"2026-07-05T10:39:49.202455+00:00"}