{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BZ4L5Q4CSKKVLAEJED6W2TRKNW","short_pith_number":"pith:BZ4L5Q4C","schema_version":"1.0","canonical_sha256":"0e78bec382929555808920fd6d4e2a6d908a943af939b72ea7b14e988119673a","source":{"kind":"arxiv","id":"2506.13497","version":1},"attestation_state":"computed","paper":{"title":"DDiT: Dynamic Resource Allocation for Diffusion Transformer Model Serving","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Cunchen Hu, Heyang Huang, Jiaqi Zhu, Liangliang Xu, Sa Wang, Sun Ninghui, Tianwei Zhang, Yizhou Shan, Yungang Bao, Ziyuan Gao","submitted_at":"2025-06-16T13:54:41Z","abstract_excerpt":"The Text-to-Video (T2V) model aims to generate dynamic and expressive videos from textual prompts. The generation pipeline typically involves multiple modules, such as language encoder, Diffusion Transformer (DiT), and Variational Autoencoders (VAE). Existing serving systems often rely on monolithic model deployment, while overlooking the distinct characteristics of each module, leading to inefficient GPU utilization. In addition, DiT exhibits varying performance gains across different resolutions and degrees of parallelism, and significant optimization potential remains unexplored. To address"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.13497","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2025-06-16T13:54:41Z","cross_cats_sorted":[],"title_canon_sha256":"f158bcc4a226c5876b8e2fa7c232dd8dd5cb719f7aabf80bccc01bf91180b339","abstract_canon_sha256":"58a47bed0f875aaf8984fdb3376da83e80c4dc005ee2de1eb63af3e6cd532d11"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:21.042156Z","signature_b64":"AoP7ccULVtyjpWnUowXfw4kgQ4He1Dj+fBHLs5oKF9xFKmlQfUWFf+2Ni+FMZ3pmM1M4ISsVtfhFfWGXzvpIBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0e78bec382929555808920fd6d4e2a6d908a943af939b72ea7b14e988119673a","last_reissued_at":"2026-07-05T11:22:21.041646Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:21.041646Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DDiT: Dynamic Resource Allocation for Diffusion Transformer Model Serving","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Cunchen Hu, Heyang Huang, Jiaqi Zhu, Liangliang Xu, Sa Wang, Sun Ninghui, Tianwei Zhang, Yizhou Shan, Yungang Bao, Ziyuan Gao","submitted_at":"2025-06-16T13:54:41Z","abstract_excerpt":"The Text-to-Video (T2V) model aims to generate dynamic and expressive videos from textual prompts. The generation pipeline typically involves multiple modules, such as language encoder, Diffusion Transformer (DiT), and Variational Autoencoders (VAE). Existing serving systems often rely on monolithic model deployment, while overlooking the distinct characteristics of each module, leading to inefficient GPU utilization. In addition, DiT exhibits varying performance gains across different resolutions and degrees of parallelism, and significant optimization potential remains unexplored. To address"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.13497","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.13497/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.13497","created_at":"2026-07-05T11:22:21.041704+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.13497v1","created_at":"2026-07-05T11:22:21.041704+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.13497","created_at":"2026-07-05T11:22:21.041704+00:00"},{"alias_kind":"pith_short_12","alias_value":"BZ4L5Q4CSKKV","created_at":"2026-07-05T11:22:21.041704+00:00"},{"alias_kind":"pith_short_16","alias_value":"BZ4L5Q4CSKKVLAEJ","created_at":"2026-07-05T11:22:21.041704+00:00"},{"alias_kind":"pith_short_8","alias_value":"BZ4L5Q4C","created_at":"2026-07-05T11:22:21.041704+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.14098","citing_title":"Cornfigurator: Automated Planning for Any-to-Any Multimodal Model Serving","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18109","citing_title":"TempoNet: Slack-Quantized Transformer-Guided Reinforcement Scheduler for Adaptive Deadline-Centric Real-Time Dispatchs","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04335","citing_title":"GENSERVE: Efficient Co-Serving of Heterogeneous Diffusion Model Workloads","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BZ4L5Q4CSKKVLAEJED6W2TRKNW","json":"https://pith.science/pith/BZ4L5Q4CSKKVLAEJED6W2TRKNW.json","graph_json":"https://pith.science/api/pith-number/BZ4L5Q4CSKKVLAEJED6W2TRKNW/graph.json","events_json":"https://pith.science/api/pith-number/BZ4L5Q4CSKKVLAEJED6W2TRKNW/events.json","paper":"https://pith.science/paper/BZ4L5Q4C"},"agent_actions":{"view_html":"https://pith.science/pith/BZ4L5Q4CSKKVLAEJED6W2TRKNW","download_json":"https://pith.science/pith/BZ4L5Q4CSKKVLAEJED6W2TRKNW.json","view_paper":"https://pith.science/paper/BZ4L5Q4C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.13497&json=true","fetch_graph":"https://pith.science/api/pith-number/BZ4L5Q4CSKKVLAEJED6W2TRKNW/graph.json","fetch_events":"https://pith.science/api/pith-number/BZ4L5Q4CSKKVLAEJED6W2TRKNW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BZ4L5Q4CSKKVLAEJED6W2TRKNW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BZ4L5Q4CSKKVLAEJED6W2TRKNW/action/storage_attestation","attest_author":"https://pith.science/pith/BZ4L5Q4CSKKVLAEJED6W2TRKNW/action/author_attestation","sign_citation":"https://pith.science/pith/BZ4L5Q4CSKKVLAEJED6W2TRKNW/action/citation_signature","submit_replication":"https://pith.science/pith/BZ4L5Q4CSKKVLAEJED6W2TRKNW/action/replication_record"}},"created_at":"2026-07-05T11:22:21.041704+00:00","updated_at":"2026-07-05T11:22:21.041704+00:00"}