{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NQOSFHZYWRTND2RS7YEDS7AXVV","short_pith_number":"pith:NQOSFHZY","schema_version":"1.0","canonical_sha256":"6c1d229f38b466d1ea32fe08397c17ad65c4555c2874d51148674fb24d94e99d","source":{"kind":"arxiv","id":"2406.08115","version":1},"attestation_state":"computed","paper":{"title":"Resource Allocation and Workload Scheduling for Large-Scale Distributed Deep Learning: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Chengming Li, Feng Liang, Haifeng Lu, Victor C. M. Leung, Xiping Hu, Yanyi Guo, Zhen Zhang","submitted_at":"2024-06-12T11:51:44Z","abstract_excerpt":"With rapidly increasing distributed deep learning workloads in large-scale data centers, efficient distributed deep learning framework strategies for resource allocation and workload scheduling have become the key to high-performance deep learning. The large-scale environment with large volumes of datasets, models, and computational and communication resources raises various unique challenges for resource allocation and workload scheduling in distributed deep learning, such as scheduling complexity, resource and workload heterogeneity, and fault tolerance. To uncover these challenges and corre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.08115","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2024-06-12T11:51:44Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a3eb440ea5fdae3842746118d7b9a1074cabf80e087764dda11d1aee0cf00fb5","abstract_canon_sha256":"ce11ac6d88ca3f30b5a20ca04a35ad717b3654db78f958b76d8fb4554d99cd9b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:30:54.908594Z","signature_b64":"tushv8pC5I4tcz+5ZzXUyad3bR3vuA+TYfqLBx7q1YO7YmeTOozgh8TO1Cn1u4sAhAHeE/ze8IVBXjEoN1/uDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6c1d229f38b466d1ea32fe08397c17ad65c4555c2874d51148674fb24d94e99d","last_reissued_at":"2026-07-05T08:30:54.908085Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:30:54.908085Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Resource Allocation and Workload Scheduling for Large-Scale Distributed Deep Learning: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Chengming Li, Feng Liang, Haifeng Lu, Victor C. M. Leung, Xiping Hu, Yanyi Guo, Zhen Zhang","submitted_at":"2024-06-12T11:51:44Z","abstract_excerpt":"With rapidly increasing distributed deep learning workloads in large-scale data centers, efficient distributed deep learning framework strategies for resource allocation and workload scheduling have become the key to high-performance deep learning. The large-scale environment with large volumes of datasets, models, and computational and communication resources raises various unique challenges for resource allocation and workload scheduling in distributed deep learning, such as scheduling complexity, resource and workload heterogeneity, and fault tolerance. To uncover these challenges and corre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.08115","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.08115/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.08115","created_at":"2026-07-05T08:30:54.908141+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.08115v1","created_at":"2026-07-05T08:30:54.908141+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.08115","created_at":"2026-07-05T08:30:54.908141+00:00"},{"alias_kind":"pith_short_12","alias_value":"NQOSFHZYWRTN","created_at":"2026-07-05T08:30:54.908141+00:00"},{"alias_kind":"pith_short_16","alias_value":"NQOSFHZYWRTND2RS","created_at":"2026-07-05T08:30:54.908141+00:00"},{"alias_kind":"pith_short_8","alias_value":"NQOSFHZY","created_at":"2026-07-05T08:30:54.908141+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22175","citing_title":"StickyInvoc: Rethinking Task Models for High-throughput Workflows in the LLM Era","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NQOSFHZYWRTND2RS7YEDS7AXVV","json":"https://pith.science/pith/NQOSFHZYWRTND2RS7YEDS7AXVV.json","graph_json":"https://pith.science/api/pith-number/NQOSFHZYWRTND2RS7YEDS7AXVV/graph.json","events_json":"https://pith.science/api/pith-number/NQOSFHZYWRTND2RS7YEDS7AXVV/events.json","paper":"https://pith.science/paper/NQOSFHZY"},"agent_actions":{"view_html":"https://pith.science/pith/NQOSFHZYWRTND2RS7YEDS7AXVV","download_json":"https://pith.science/pith/NQOSFHZYWRTND2RS7YEDS7AXVV.json","view_paper":"https://pith.science/paper/NQOSFHZY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.08115&json=true","fetch_graph":"https://pith.science/api/pith-number/NQOSFHZYWRTND2RS7YEDS7AXVV/graph.json","fetch_events":"https://pith.science/api/pith-number/NQOSFHZYWRTND2RS7YEDS7AXVV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NQOSFHZYWRTND2RS7YEDS7AXVV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NQOSFHZYWRTND2RS7YEDS7AXVV/action/storage_attestation","attest_author":"https://pith.science/pith/NQOSFHZYWRTND2RS7YEDS7AXVV/action/author_attestation","sign_citation":"https://pith.science/pith/NQOSFHZYWRTND2RS7YEDS7AXVV/action/citation_signature","submit_replication":"https://pith.science/pith/NQOSFHZYWRTND2RS7YEDS7AXVV/action/replication_record"}},"created_at":"2026-07-05T08:30:54.908141+00:00","updated_at":"2026-07-05T08:30:54.908141+00:00"}