{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LYZRYSZOAPC753H4MHZNKYXWFD","short_pith_number":"pith:LYZRYSZO","schema_version":"1.0","canonical_sha256":"5e331c4b2e03c5feecfc61f2d562f628fe7d5fd9dd12b449328ff714783eeffd","source":{"kind":"arxiv","id":"2406.08756","version":4},"attestation_state":"computed","paper":{"title":"Optimizing Large Model Training through Overlapped Activation Recomputation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Gang Chen, Kexin Huang, Ping Chen, Shuibing He, Siling Yang, Weijian Chen, Wenjie Zhang, Xuan Zhan, Yanlong Yin, Yingjie Gu, Yi Zheng, Zhefeng Wang, Zhuwei Peng","submitted_at":"2024-06-13T02:31:36Z","abstract_excerpt":"Large model training often uses recomputation to alleviate memory pressure and pipelines to exploit the parallelism of data, tensors, and devices. However, existing recomputation approaches may incur high overhead when training real-world models, as they are executed on demand in the critical training path. In this paper, we present Lynx, a new recomputation framework to reduce overhead by overlapping recomputation with communication in training pipelines. To reduce the large search space for recomputation strategies, we propose a heuristic-based recomputation scheduling algorithm, which is ba"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.08756","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2024-06-13T02:31:36Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"9f1425ad21096776521bafc7bd32f6ff8118c8384bbc588bdeeffb08b4bd1944","abstract_canon_sha256":"f40a3504b73250fc9d34ffe6f8418508970572dbf86e95cda01e85d4e0615de0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:40:34.072131Z","signature_b64":"QTV+fUNw1DO+RG9kurLsrUGl2AhHT+20/S39HHMVGnrHolm1KEsPBnMVNXelzR9Xu/6DQKLiUtVPMT5LObEbDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5e331c4b2e03c5feecfc61f2d562f628fe7d5fd9dd12b449328ff714783eeffd","last_reissued_at":"2026-07-05T10:40:34.071638Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:40:34.071638Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Optimizing Large Model Training through Overlapped Activation Recomputation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Gang Chen, Kexin Huang, Ping Chen, Shuibing He, Siling Yang, Weijian Chen, Wenjie Zhang, Xuan Zhan, Yanlong Yin, Yingjie Gu, Yi Zheng, Zhefeng Wang, Zhuwei Peng","submitted_at":"2024-06-13T02:31:36Z","abstract_excerpt":"Large model training often uses recomputation to alleviate memory pressure and pipelines to exploit the parallelism of data, tensors, and devices. However, existing recomputation approaches may incur high overhead when training real-world models, as they are executed on demand in the critical training path. In this paper, we present Lynx, a new recomputation framework to reduce overhead by overlapping recomputation with communication in training pipelines. To reduce the large search space for recomputation strategies, we propose a heuristic-based recomputation scheduling algorithm, which is ba"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.08756","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.08756/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.08756","created_at":"2026-07-05T10:40:34.071697+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.08756v4","created_at":"2026-07-05T10:40:34.071697+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.08756","created_at":"2026-07-05T10:40:34.071697+00:00"},{"alias_kind":"pith_short_12","alias_value":"LYZRYSZOAPC7","created_at":"2026-07-05T10:40:34.071697+00:00"},{"alias_kind":"pith_short_16","alias_value":"LYZRYSZOAPC753H4","created_at":"2026-07-05T10:40:34.071697+00:00"},{"alias_kind":"pith_short_8","alias_value":"LYZRYSZO","created_at":"2026-07-05T10:40:34.071697+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2409.01143","citing_title":"HexiScale: Facilitating Large Language Model Training over Heterogeneous Hardware","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LYZRYSZOAPC753H4MHZNKYXWFD","json":"https://pith.science/pith/LYZRYSZOAPC753H4MHZNKYXWFD.json","graph_json":"https://pith.science/api/pith-number/LYZRYSZOAPC753H4MHZNKYXWFD/graph.json","events_json":"https://pith.science/api/pith-number/LYZRYSZOAPC753H4MHZNKYXWFD/events.json","paper":"https://pith.science/paper/LYZRYSZO"},"agent_actions":{"view_html":"https://pith.science/pith/LYZRYSZOAPC753H4MHZNKYXWFD","download_json":"https://pith.science/pith/LYZRYSZOAPC753H4MHZNKYXWFD.json","view_paper":"https://pith.science/paper/LYZRYSZO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.08756&json=true","fetch_graph":"https://pith.science/api/pith-number/LYZRYSZOAPC753H4MHZNKYXWFD/graph.json","fetch_events":"https://pith.science/api/pith-number/LYZRYSZOAPC753H4MHZNKYXWFD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LYZRYSZOAPC753H4MHZNKYXWFD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LYZRYSZOAPC753H4MHZNKYXWFD/action/storage_attestation","attest_author":"https://pith.science/pith/LYZRYSZOAPC753H4MHZNKYXWFD/action/author_attestation","sign_citation":"https://pith.science/pith/LYZRYSZOAPC753H4MHZNKYXWFD/action/citation_signature","submit_replication":"https://pith.science/pith/LYZRYSZOAPC753H4MHZNKYXWFD/action/replication_record"}},"created_at":"2026-07-05T10:40:34.071697+00:00","updated_at":"2026-07-05T10:40:34.071697+00:00"}