{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:A7BULS5IB6IJYP2IHCSBCFHBSR","short_pith_number":"pith:A7BULS5I","schema_version":"1.0","canonical_sha256":"07c345cba80f909c3f4838a41114e19472fa9c30458030ac08ce59eb2db43623","source":{"kind":"arxiv","id":"2306.09305","version":2},"attestation_state":"computed","paper":{"title":"Fast Training of Diffusion Models with Masked Transformers","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Anima Anandkumar, Arash Vahdat, Hongkai Zheng, Weili Nie","submitted_at":"2023-06-15T17:38:48Z","abstract_excerpt":"We propose an efficient approach to train large diffusion models with masked transformers. While masked transformers have been extensively explored for representation learning, their application to generative learning is less explored in the vision domain. Our work is the first to exploit masked training to reduce the training cost of diffusion models significantly. Specifically, we randomly mask out a high proportion (e.g., 50%) of patches in diffused input images during training. For masked training, we introduce an asymmetric encoder-decoder architecture consisting of a transformer encoder "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.09305","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2023-06-15T17:38:48Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"a1b217dfdc829c82e37447ceaef4b7e1b1ce29d298ae401ed98fa5bef9ecbd28","abstract_canon_sha256":"08568984644ab004aca7561fc9e4aeef28871c7caa36db490cd913a74f766991"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:52:10.452141Z","signature_b64":"ohOYWVbegowKz7THCCnSZ2s1uHFGgLb7IryBpySXxygMgI9oSwJTqeERsW1nq72nUoYEK3oULRNXgmaUN3fxCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"07c345cba80f909c3f4838a41114e19472fa9c30458030ac08ce59eb2db43623","last_reissued_at":"2026-07-05T07:52:10.451703Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:52:10.451703Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fast Training of Diffusion Models with Masked Transformers","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Anima Anandkumar, Arash Vahdat, Hongkai Zheng, Weili Nie","submitted_at":"2023-06-15T17:38:48Z","abstract_excerpt":"We propose an efficient approach to train large diffusion models with masked transformers. While masked transformers have been extensively explored for representation learning, their application to generative learning is less explored in the vision domain. Our work is the first to exploit masked training to reduce the training cost of diffusion models significantly. Specifically, we randomly mask out a high proportion (e.g., 50%) of patches in diffused input images during training. For masked training, we introduce an asymmetric encoder-decoder architecture consisting of a transformer encoder "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.09305","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.09305/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.09305","created_at":"2026-07-05T07:52:10.451763+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.09305v2","created_at":"2026-07-05T07:52:10.451763+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.09305","created_at":"2026-07-05T07:52:10.451763+00:00"},{"alias_kind":"pith_short_12","alias_value":"A7BULS5IB6IJ","created_at":"2026-07-05T07:52:10.451763+00:00"},{"alias_kind":"pith_short_16","alias_value":"A7BULS5IB6IJYP2I","created_at":"2026-07-05T07:52:10.451763+00:00"},{"alias_kind":"pith_short_8","alias_value":"A7BULS5I","created_at":"2026-07-05T07:52:10.451763+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24888","citing_title":"DiffusionBench: On Holistic Evaluation of Diffusion Transformers","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02508","citing_title":"From SRA to Self-Flow: Data Augmentation or Self-Supervision?","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11096","citing_title":"IDEAL: In-DEpth ALignment Makes A Discrete Representation AutoEncoder","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08788","citing_title":"MaskAlign: Token-Subset Representation Alignment for Efficient Diffusion Training","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18267","citing_title":"SRC-Flow: Compact Semantic Representations Enable Normalizing Flows for Image Generation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27760","citing_title":"PixelU: A U-Shaped Transformer for Efficient End-to-End Pixel Diffusion","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06886","citing_title":"Prompt Reinjection: Alleviating Prompt Forgetting in Multimodal Diffusion Transformers for Text-to-Image Generation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17759","citing_title":"FrequencyBooster: Full-Frequency Modeling for High-Fidelity Pixel Diffusion","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18267","citing_title":"SRC-Flow: Compact Semantic Representations Enable Normalizing Flows for Image Generation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18390","citing_title":"Vision Foundation Models as Generalist Tokenizers for Image Generation","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16949","citing_title":"Beyond Point-Wise Matching: Structural Representation Alignment for Accelerating Diffusion Transformers","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2506.13763","citing_title":"Diagnosing and Improving Diffusion Models by Estimating the Optimal Loss Value","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18457","citing_title":"VFM-VAE: Vision Foundation Models Can Be Good Tokenizers for Latent Diffusion Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2310.05737","citing_title":"Language Model Beats Diffusion -- Tokenizer is Key to Visual Generation","ref_index":241,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10790","citing_title":"Elucidating Representation Degradation Problem in Diffusion Model Training","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00503","citing_title":"End-to-End Autoregressive Image Generation with 1D Semantic Tokenizer","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09168","citing_title":"ELT: Elastic Looped Transformers for Visual Generation","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07397","citing_title":"Data Warmup: Complexity-Aware Curricula for Efficient Diffusion Training","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19141","citing_title":"Denoising, Fast and Slow: Difficulty-Aware Adaptive Sampling for Image Generation","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A7BULS5IB6IJYP2IHCSBCFHBSR","json":"https://pith.science/pith/A7BULS5IB6IJYP2IHCSBCFHBSR.json","graph_json":"https://pith.science/api/pith-number/A7BULS5IB6IJYP2IHCSBCFHBSR/graph.json","events_json":"https://pith.science/api/pith-number/A7BULS5IB6IJYP2IHCSBCFHBSR/events.json","paper":"https://pith.science/paper/A7BULS5I"},"agent_actions":{"view_html":"https://pith.science/pith/A7BULS5IB6IJYP2IHCSBCFHBSR","download_json":"https://pith.science/pith/A7BULS5IB6IJYP2IHCSBCFHBSR.json","view_paper":"https://pith.science/paper/A7BULS5I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.09305&json=true","fetch_graph":"https://pith.science/api/pith-number/A7BULS5IB6IJYP2IHCSBCFHBSR/graph.json","fetch_events":"https://pith.science/api/pith-number/A7BULS5IB6IJYP2IHCSBCFHBSR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A7BULS5IB6IJYP2IHCSBCFHBSR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A7BULS5IB6IJYP2IHCSBCFHBSR/action/storage_attestation","attest_author":"https://pith.science/pith/A7BULS5IB6IJYP2IHCSBCFHBSR/action/author_attestation","sign_citation":"https://pith.science/pith/A7BULS5IB6IJYP2IHCSBCFHBSR/action/citation_signature","submit_replication":"https://pith.science/pith/A7BULS5IB6IJYP2IHCSBCFHBSR/action/replication_record"}},"created_at":"2026-07-05T07:52:10.451763+00:00","updated_at":"2026-07-05T07:52:10.451763+00:00"}