{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VKCFEA5MRYWJTL6HIIX4PZZG7B","short_pith_number":"pith:VKCFEA5M","schema_version":"1.0","canonical_sha256":"aa845203ac8e2c99afc7422fc7e726f84e3cfdd898d7b804e8fc7d1b1e2693f0","source":{"kind":"arxiv","id":"2407.20143","version":4},"attestation_state":"computed","paper":{"title":"ByteCheckpoint: A Unified Checkpointing System for Large Foundation Model Development","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Borui Wan, Chuan Wu, Haibin Lin, Junda Zhang, Menghan Yu, Mingji Han, Mofan Zhang, Xin Liu, Yanghua Peng, Yiyao Sheng, Zhichao Lai, Zuquan Song","submitted_at":"2024-07-29T16:18:20Z","abstract_excerpt":"Checkpointing to preserve training states is crucial during the development of Large Foundation Models (LFMs), for training resumption upon various failures or changes in GPU resources and parallelism configurations. In addition, saved checkpoints are dispatched to evaluation tasks or transferred across different training stages (e.g., from pre-training to post-training). All these scenarios require resharding distributed checkpoints from one parallelism to another. In production environments, different LFMs are trained with various frameworks and storage backends, depending on model sizes and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.20143","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-07-29T16:18:20Z","cross_cats_sorted":[],"title_canon_sha256":"d366d54f186c917c2210ca7aa75708293f16c2b47046afa66a8fc2090864fecf","abstract_canon_sha256":"202da417b27d2de07e1fd9b533bf22d1ff514f96131841584dcccc9664d763ac"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:43:00.486656Z","signature_b64":"cCv6ckejJdPfiW3J15ZUUq1mf3PHEyuM8gC/4f3q48bz6jaDFc062n/RgE3Hiu8RR2cdMJ62/83IbIHVNaK2BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aa845203ac8e2c99afc7422fc7e726f84e3cfdd898d7b804e8fc7d1b1e2693f0","last_reissued_at":"2026-07-05T10:43:00.486144Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:43:00.486144Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ByteCheckpoint: A Unified Checkpointing System for Large Foundation Model Development","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Borui Wan, Chuan Wu, Haibin Lin, Junda Zhang, Menghan Yu, Mingji Han, Mofan Zhang, Xin Liu, Yanghua Peng, Yiyao Sheng, Zhichao Lai, Zuquan Song","submitted_at":"2024-07-29T16:18:20Z","abstract_excerpt":"Checkpointing to preserve training states is crucial during the development of Large Foundation Models (LFMs), for training resumption upon various failures or changes in GPU resources and parallelism configurations. In addition, saved checkpoints are dispatched to evaluation tasks or transferred across different training stages (e.g., from pre-training to post-training). All these scenarios require resharding distributed checkpoints from one parallelism to another. In production environments, different LFMs are trained with various frameworks and storage backends, depending on model sizes and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.20143","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.20143/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.20143","created_at":"2026-07-05T10:43:00.486209+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.20143v4","created_at":"2026-07-05T10:43:00.486209+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.20143","created_at":"2026-07-05T10:43:00.486209+00:00"},{"alias_kind":"pith_short_12","alias_value":"VKCFEA5MRYWJ","created_at":"2026-07-05T10:43:00.486209+00:00"},{"alias_kind":"pith_short_16","alias_value":"VKCFEA5MRYWJTL6H","created_at":"2026-07-05T10:43:00.486209+00:00"},{"alias_kind":"pith_short_8","alias_value":"VKCFEA5M","created_at":"2026-07-05T10:43:00.486209+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01646","citing_title":"PHOENIX: Resilient LLM Training with Hot-Swapping via Zero-Overhead Checkpoint","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2504.09844","citing_title":"MegaScale-Data: Scaling Dataloader for Multisource Large Foundation Model Training","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08962","citing_title":"MegaScale-Omni: A Hyper-Scale, Workload-Resilient System for MultiModal LLM Training in Production","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2505.07062","citing_title":"Seed1.5-VL Technical Report","ref_index":137,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VKCFEA5MRYWJTL6HIIX4PZZG7B","json":"https://pith.science/pith/VKCFEA5MRYWJTL6HIIX4PZZG7B.json","graph_json":"https://pith.science/api/pith-number/VKCFEA5MRYWJTL6HIIX4PZZG7B/graph.json","events_json":"https://pith.science/api/pith-number/VKCFEA5MRYWJTL6HIIX4PZZG7B/events.json","paper":"https://pith.science/paper/VKCFEA5M"},"agent_actions":{"view_html":"https://pith.science/pith/VKCFEA5MRYWJTL6HIIX4PZZG7B","download_json":"https://pith.science/pith/VKCFEA5MRYWJTL6HIIX4PZZG7B.json","view_paper":"https://pith.science/paper/VKCFEA5M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.20143&json=true","fetch_graph":"https://pith.science/api/pith-number/VKCFEA5MRYWJTL6HIIX4PZZG7B/graph.json","fetch_events":"https://pith.science/api/pith-number/VKCFEA5MRYWJTL6HIIX4PZZG7B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VKCFEA5MRYWJTL6HIIX4PZZG7B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VKCFEA5MRYWJTL6HIIX4PZZG7B/action/storage_attestation","attest_author":"https://pith.science/pith/VKCFEA5MRYWJTL6HIIX4PZZG7B/action/author_attestation","sign_citation":"https://pith.science/pith/VKCFEA5MRYWJTL6HIIX4PZZG7B/action/citation_signature","submit_replication":"https://pith.science/pith/VKCFEA5MRYWJTL6HIIX4PZZG7B/action/replication_record"}},"created_at":"2026-07-05T10:43:00.486209+00:00","updated_at":"2026-07-05T10:43:00.486209+00:00"}