{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7RZYR5LDJMG4WZA3CYU3QDUDCQ","short_pith_number":"pith:7RZYR5LD","schema_version":"1.0","canonical_sha256":"fc7388f5634b0dcb641b1629b80e8314287d04c65c6eab67ac383239cd2f082c","source":{"kind":"arxiv","id":"2502.03444","version":2},"attestation_state":"computed","paper":{"title":"Masked Autoencoders Are Effective Tokenizers for Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bhiksha Raj, Difan Zou, Fangyi Chen, Hao Chen, Jindong Wang, Xiang Li, Yidong Wang, Yujin Han, Ze Wang, Zicheng Liu","submitted_at":"2025-02-05T18:42:04Z","abstract_excerpt":"Recent advances in latent diffusion models have demonstrated their effectiveness for high-resolution image synthesis. However, the properties of the latent space from tokenizer for better learning and generation of diffusion models remain under-explored. Theoretically and empirically, we find that improved generation quality is closely tied to the latent distributions with better structure, such as the ones with fewer Gaussian Mixture modes and more discriminative features. Motivated by these insights, we propose MAETok, an autoencoder (AE) leveraging mask modeling to learn semantically rich l"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.03444","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-05T18:42:04Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"31445b310173db6a8c4541d7068b4ba29e641d1058a513062cc5f8988306eb4c","abstract_canon_sha256":"e2ea62fcf113b43569ef9c20c0460e622a666ccb519e18f3bf712bbb265b28ed"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:25.448707Z","signature_b64":"jsg4qQfI6wEJjIxiFOdNAAclRDiMzv+kxaIckyoWQjZqJAHw391E4mOlOQh1/tsab8VjBmWam32y/kUSrn/KCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fc7388f5634b0dcb641b1629b80e8314287d04c65c6eab67ac383239cd2f082c","last_reissued_at":"2026-07-05T11:12:25.448182Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:25.448182Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Masked Autoencoders Are Effective Tokenizers for Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bhiksha Raj, Difan Zou, Fangyi Chen, Hao Chen, Jindong Wang, Xiang Li, Yidong Wang, Yujin Han, Ze Wang, Zicheng Liu","submitted_at":"2025-02-05T18:42:04Z","abstract_excerpt":"Recent advances in latent diffusion models have demonstrated their effectiveness for high-resolution image synthesis. However, the properties of the latent space from tokenizer for better learning and generation of diffusion models remain under-explored. Theoretically and empirically, we find that improved generation quality is closely tied to the latent distributions with better structure, such as the ones with fewer Gaussian Mixture modes and more discriminative features. Motivated by these insights, we propose MAETok, an autoencoder (AE) leveraging mask modeling to learn semantically rich l"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.03444","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.03444/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.03444","created_at":"2026-07-05T11:12:25.448249+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.03444v2","created_at":"2026-07-05T11:12:25.448249+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.03444","created_at":"2026-07-05T11:12:25.448249+00:00"},{"alias_kind":"pith_short_12","alias_value":"7RZYR5LDJMG4","created_at":"2026-07-05T11:12:25.448249+00:00"},{"alias_kind":"pith_short_16","alias_value":"7RZYR5LDJMG4WZA3","created_at":"2026-07-05T11:12:25.448249+00:00"},{"alias_kind":"pith_short_8","alias_value":"7RZYR5LD","created_at":"2026-07-05T11:12:25.448249+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26016","citing_title":"MIMFlow: Integrating Masked Image Modeling with Normalizing Flows for End-to-End Image Generation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19651","citing_title":"BrainG3N: A Dual-Purpose Tokenizer for Controllable 3D Brain MRI Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27696","citing_title":"Structure over Pixels: Learning Variable-Length Visual Programs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24885","citing_title":"VibeToken: Scaling 1D Image Tokenizers and Autoregressive Models for Dynamic Resolution Generations","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07915","citing_title":"What Matters for Diffusion-Friendly Latent Manifold? Prior-Aligned Autoencoders for Latent Diffusion","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7RZYR5LDJMG4WZA3CYU3QDUDCQ","json":"https://pith.science/pith/7RZYR5LDJMG4WZA3CYU3QDUDCQ.json","graph_json":"https://pith.science/api/pith-number/7RZYR5LDJMG4WZA3CYU3QDUDCQ/graph.json","events_json":"https://pith.science/api/pith-number/7RZYR5LDJMG4WZA3CYU3QDUDCQ/events.json","paper":"https://pith.science/paper/7RZYR5LD"},"agent_actions":{"view_html":"https://pith.science/pith/7RZYR5LDJMG4WZA3CYU3QDUDCQ","download_json":"https://pith.science/pith/7RZYR5LDJMG4WZA3CYU3QDUDCQ.json","view_paper":"https://pith.science/paper/7RZYR5LD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.03444&json=true","fetch_graph":"https://pith.science/api/pith-number/7RZYR5LDJMG4WZA3CYU3QDUDCQ/graph.json","fetch_events":"https://pith.science/api/pith-number/7RZYR5LDJMG4WZA3CYU3QDUDCQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7RZYR5LDJMG4WZA3CYU3QDUDCQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7RZYR5LDJMG4WZA3CYU3QDUDCQ/action/storage_attestation","attest_author":"https://pith.science/pith/7RZYR5LDJMG4WZA3CYU3QDUDCQ/action/author_attestation","sign_citation":"https://pith.science/pith/7RZYR5LDJMG4WZA3CYU3QDUDCQ/action/citation_signature","submit_replication":"https://pith.science/pith/7RZYR5LDJMG4WZA3CYU3QDUDCQ/action/replication_record"}},"created_at":"2026-07-05T11:12:25.448249+00:00","updated_at":"2026-07-05T11:12:25.448249+00:00"}