{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:JB2RKQ6ZL7ACFOIVDX2TWDPMMZ","short_pith_number":"pith:JB2RKQ6Z","schema_version":"1.0","canonical_sha256":"48751543d95fc022b9151df53b0dec66710b2ba070cae4de6ee01d55adc38298","source":{"kind":"arxiv","id":"2205.05198","version":1},"attestation_state":"computed","paper":{"title":"Reducing Activation Recomputation in Large Transformer Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Bryan Catanzaro, Jared Casper, Lawrence McAfee, Michael Andersch, Mohammad Shoeybi, Sangkug Lym, Vijay Korthikanti","submitted_at":"2022-05-10T22:40:17Z","abstract_excerpt":"Training large transformer models is one of the most important computational challenges of modern AI. In this paper, we show how to significantly accelerate training of large transformer models by reducing activation recomputation. Activation recomputation is commonly used to work around memory capacity constraints. Rather than storing activations for backpropagation, they are traditionally recomputed, which saves memory but adds redundant compute. In this work, we show most of this redundant compute is unnecessary because we can reduce memory consumption sufficiently without it. We present tw"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.05198","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-05-10T22:40:17Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"ca9e4db5fbb76544c2b8651a3dc7fdda471bf9f3338b303163efd65d58348b3d","abstract_canon_sha256":"e4ace58f724b13ae2426e6e7b63b73fef3ce98753d51a34ead639bed3dd4d40c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:22:23.156461Z","signature_b64":"q1bCacqNNGJlk8X4XihNOF+Dpq4Z+bJyCJvMushOJaRaR2EhdPsVz3S+L6DbGCYx0guj77PmP7Tykk2sQatIAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"48751543d95fc022b9151df53b0dec66710b2ba070cae4de6ee01d55adc38298","last_reissued_at":"2026-07-05T04:22:23.156018Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:22:23.156018Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reducing Activation Recomputation in Large Transformer Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Bryan Catanzaro, Jared Casper, Lawrence McAfee, Michael Andersch, Mohammad Shoeybi, Sangkug Lym, Vijay Korthikanti","submitted_at":"2022-05-10T22:40:17Z","abstract_excerpt":"Training large transformer models is one of the most important computational challenges of modern AI. In this paper, we show how to significantly accelerate training of large transformer models by reducing activation recomputation. Activation recomputation is commonly used to work around memory capacity constraints. Rather than storing activations for backpropagation, they are traditionally recomputed, which saves memory but adds redundant compute. In this work, we show most of this redundant compute is unnecessary because we can reduce memory consumption sufficiently without it. We present tw"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.05198","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.05198/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.05198","created_at":"2026-07-05T04:22:23.156076+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.05198v1","created_at":"2026-07-05T04:22:23.156076+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.05198","created_at":"2026-07-05T04:22:23.156076+00:00"},{"alias_kind":"pith_short_12","alias_value":"JB2RKQ6ZL7AC","created_at":"2026-07-05T04:22:23.156076+00:00"},{"alias_kind":"pith_short_16","alias_value":"JB2RKQ6ZL7ACFOIV","created_at":"2026-07-05T04:22:23.156076+00:00"},{"alias_kind":"pith_short_8","alias_value":"JB2RKQ6Z","created_at":"2026-07-05T04:22:23.156076+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08167","citing_title":"Explaining Data Mixing Scaling Laws","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2503.20314","citing_title":"Wan: Open and Advanced Large-Scale Video Generative Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2504.09844","citing_title":"MegaScale-Data: Scaling Dataloader for Multisource Large Foundation Model Training","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2504.10013","citing_title":"Training LLMs on HPC Systems: Best Practices from the OpenGPT-X Project","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20799","citing_title":"Instant GPU Efficiency Visibility at Fleet Scale","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2508.21613","citing_title":"Chameleon: Adaptive Fault Tolerance for Distributed Training via Real-time Policy Selection","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2509.05276","citing_title":"SpikingBrain: Spiking Brain-inspired Large Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21275","citing_title":"InfiniPipe: Elastic Pipeline Parallelism for Efficient Variable-Length Long-Context LLM Training","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2406.07887","citing_title":"An Empirical Study of Mamba-based Language Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2512.20856","citing_title":"NVIDIA Nemotron 3: Efficient and Open Intelligence","ref_index":128,"is_internal_anchor":false},{"citing_arxiv_id":"2303.17564","citing_title":"BloombergGPT: A Large Language Model for Finance","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2310.01889","citing_title":"Ring Attention with Blockwise Transformers for Near-Infinite Context","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27085","citing_title":"Efficient Training on Multiple Consumer GPUs with RoundPipe","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2304.11277","citing_title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08962","citing_title":"MegaScale-Omni: A Hyper-Scale, Workload-Resilient System for MultiModal LLM Training in Production","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05049","citing_title":"Piper: Efficient Large-Scale MoE Training via Resource Modeling and Pipelined Hybrid Parallelism","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2303.18223","citing_title":"A Survey of Large Language Models","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14561","citing_title":"CoCoDiff: Optimizing Collective Communications for Distributed Diffusion Transformer Inference Under Ulysses Sequence Parallelism","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21428","citing_title":"Decoupled DiLoCo for Resilient Distributed Pre-training","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JB2RKQ6ZL7ACFOIVDX2TWDPMMZ","json":"https://pith.science/pith/JB2RKQ6ZL7ACFOIVDX2TWDPMMZ.json","graph_json":"https://pith.science/api/pith-number/JB2RKQ6ZL7ACFOIVDX2TWDPMMZ/graph.json","events_json":"https://pith.science/api/pith-number/JB2RKQ6ZL7ACFOIVDX2TWDPMMZ/events.json","paper":"https://pith.science/paper/JB2RKQ6Z"},"agent_actions":{"view_html":"https://pith.science/pith/JB2RKQ6ZL7ACFOIVDX2TWDPMMZ","download_json":"https://pith.science/pith/JB2RKQ6ZL7ACFOIVDX2TWDPMMZ.json","view_paper":"https://pith.science/paper/JB2RKQ6Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.05198&json=true","fetch_graph":"https://pith.science/api/pith-number/JB2RKQ6ZL7ACFOIVDX2TWDPMMZ/graph.json","fetch_events":"https://pith.science/api/pith-number/JB2RKQ6ZL7ACFOIVDX2TWDPMMZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JB2RKQ6ZL7ACFOIVDX2TWDPMMZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JB2RKQ6ZL7ACFOIVDX2TWDPMMZ/action/storage_attestation","attest_author":"https://pith.science/pith/JB2RKQ6ZL7ACFOIVDX2TWDPMMZ/action/author_attestation","sign_citation":"https://pith.science/pith/JB2RKQ6ZL7ACFOIVDX2TWDPMMZ/action/citation_signature","submit_replication":"https://pith.science/pith/JB2RKQ6ZL7ACFOIVDX2TWDPMMZ/action/replication_record"}},"created_at":"2026-07-05T04:22:23.156076+00:00","updated_at":"2026-07-05T04:22:23.156076+00:00"}