{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WWLGIOQUUDUPJBFUNVSOIFCWGB","short_pith_number":"pith:WWLGIOQU","schema_version":"1.0","canonical_sha256":"b596643a14a0e8f484b46d64e41456305526f4770bb8cc1018e1fedff8cbe8aa","source":{"kind":"arxiv","id":"2410.02098","version":5},"attestation_state":"computed","paper":{"title":"EC-DIT: Scaling Diffusion Transformers with Adaptive Expert-Choice Routing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Bo Dai, Bowen Zhang, Haoshuo Huang, Haotian Sun, Nan Du, Ruoming Pang, Tao Lei, Yanghao Li","submitted_at":"2024-10-02T23:39:10Z","abstract_excerpt":"Diffusion transformers have been widely adopted for text-to-image synthesis. While scaling these models up to billions of parameters shows promise, the effectiveness of scaling beyond current sizes remains underexplored and challenging. By explicitly exploiting the computational heterogeneity of image generations, we develop a new family of Mixture-of-Experts (MoE) models (EC-DIT) for diffusion transformers with expert-choice routing. EC-DIT learns to adaptively optimize the compute allocated to understand the input texts and generate the respective image patches, enabling heterogeneous comput"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.02098","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-02T23:39:10Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"d015c664c360567d15e6bda5e2c0cd3f602d31e2ce6322db527aa433f898d277","abstract_canon_sha256":"3a0fce4f4d672509c343b9302a72326b6eac26f79b3996d7db517d221b9b2512"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:23:56.740845Z","signature_b64":"LxSexfceLRaEMXTIVwsskNq4GugBvjtLOfURm2TxSUmp2UFqqa4UY3t15aEXnMYEwjJnK0ss0bo8gUHJxvhTAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b596643a14a0e8f484b46d64e41456305526f4770bb8cc1018e1fedff8cbe8aa","last_reissued_at":"2026-07-05T10:23:56.740231Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:23:56.740231Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EC-DIT: Scaling Diffusion Transformers with Adaptive Expert-Choice Routing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Bo Dai, Bowen Zhang, Haoshuo Huang, Haotian Sun, Nan Du, Ruoming Pang, Tao Lei, Yanghao Li","submitted_at":"2024-10-02T23:39:10Z","abstract_excerpt":"Diffusion transformers have been widely adopted for text-to-image synthesis. While scaling these models up to billions of parameters shows promise, the effectiveness of scaling beyond current sizes remains underexplored and challenging. By explicitly exploiting the computational heterogeneity of image generations, we develop a new family of Mixture-of-Experts (MoE) models (EC-DIT) for diffusion transformers with expert-choice routing. EC-DIT learns to adaptively optimize the compute allocated to understand the input texts and generate the respective image patches, enabling heterogeneous comput"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.02098","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.02098/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.02098","created_at":"2026-07-05T10:23:56.740300+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.02098v5","created_at":"2026-07-05T10:23:56.740300+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.02098","created_at":"2026-07-05T10:23:56.740300+00:00"},{"alias_kind":"pith_short_12","alias_value":"WWLGIOQUUDUP","created_at":"2026-07-05T10:23:56.740300+00:00"},{"alias_kind":"pith_short_16","alias_value":"WWLGIOQUUDUPJBFU","created_at":"2026-07-05T10:23:56.740300+00:00"},{"alias_kind":"pith_short_8","alias_value":"WWLGIOQU","created_at":"2026-07-05T10:23:56.740300+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26938","citing_title":"Focusing on What Matters: Saliency-Harnessing Accurate Routing for Diffusion MoE","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21033","citing_title":"MoECodec: Image Compression for joint human and machine perception via Mixture-of-Experts","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02090","citing_title":"FocusDiT: Masking Queries in Diffusion Transformers for Fine-grained Image Generation","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23893","citing_title":"Complete-muE: Optimal Hyperparameter Transfer and Scaling for MoE Models","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WWLGIOQUUDUPJBFUNVSOIFCWGB","json":"https://pith.science/pith/WWLGIOQUUDUPJBFUNVSOIFCWGB.json","graph_json":"https://pith.science/api/pith-number/WWLGIOQUUDUPJBFUNVSOIFCWGB/graph.json","events_json":"https://pith.science/api/pith-number/WWLGIOQUUDUPJBFUNVSOIFCWGB/events.json","paper":"https://pith.science/paper/WWLGIOQU"},"agent_actions":{"view_html":"https://pith.science/pith/WWLGIOQUUDUPJBFUNVSOIFCWGB","download_json":"https://pith.science/pith/WWLGIOQUUDUPJBFUNVSOIFCWGB.json","view_paper":"https://pith.science/paper/WWLGIOQU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.02098&json=true","fetch_graph":"https://pith.science/api/pith-number/WWLGIOQUUDUPJBFUNVSOIFCWGB/graph.json","fetch_events":"https://pith.science/api/pith-number/WWLGIOQUUDUPJBFUNVSOIFCWGB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WWLGIOQUUDUPJBFUNVSOIFCWGB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WWLGIOQUUDUPJBFUNVSOIFCWGB/action/storage_attestation","attest_author":"https://pith.science/pith/WWLGIOQUUDUPJBFUNVSOIFCWGB/action/author_attestation","sign_citation":"https://pith.science/pith/WWLGIOQUUDUPJBFUNVSOIFCWGB/action/citation_signature","submit_replication":"https://pith.science/pith/WWLGIOQUUDUPJBFUNVSOIFCWGB/action/replication_record"}},"created_at":"2026-07-05T10:23:56.740300+00:00","updated_at":"2026-07-05T10:23:56.740300+00:00"}