{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PDXHAZK6X6HRW5C6ZWPFNAQCJY","short_pith_number":"pith:PDXHAZK6","schema_version":"1.0","canonical_sha256":"78ee70655ebf8f1b745ecd9e5682024e339bf39e60e166b04c11d0581cdde2a1","source":{"kind":"arxiv","id":"2302.01107","version":3},"attestation_state":"computed","paper":{"title":"A Survey on Efficient Training of Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Bohan Zhuang, Chunhua Shen, Haoyu He, Jing Liu, Yuetian Weng, Zizheng Pan","submitted_at":"2023-02-02T13:58:18Z","abstract_excerpt":"Recent advances in Transformers have come with a huge requirement on computing resources, highlighting the importance of developing efficient training techniques to make Transformer training faster, at lower cost, and to higher accuracy by the efficient use of computation and memory resources. This survey provides the first systematic overview of the efficient training of Transformers, covering the recent progress in acceleration arithmetic and hardware, with a focus on the former. We analyze and compare methods that save computation and memory costs for intermediate tensors during training, t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.01107","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-02-02T13:58:18Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"587fea03314c45792d1e0802c0ae1295467cdb02d942b20ed7ad7a111e380893","abstract_canon_sha256":"66e14bf483ade2ba53761c83031ef12290a2efd2c6ade65a8dfc285899351823"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:06:57.689032Z","signature_b64":"Enl1tkik0B2ZcEJuBJ8c8NRcZYvYJDp2uF6G0jUPIK6h6Cxjqr3o96tftsuWkhIrpiA5k0G+hAICg6c/qsE1DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"78ee70655ebf8f1b745ecd9e5682024e339bf39e60e166b04c11d0581cdde2a1","last_reissued_at":"2026-07-05T06:06:57.688616Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:06:57.688616Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey on Efficient Training of Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Bohan Zhuang, Chunhua Shen, Haoyu He, Jing Liu, Yuetian Weng, Zizheng Pan","submitted_at":"2023-02-02T13:58:18Z","abstract_excerpt":"Recent advances in Transformers have come with a huge requirement on computing resources, highlighting the importance of developing efficient training techniques to make Transformer training faster, at lower cost, and to higher accuracy by the efficient use of computation and memory resources. This survey provides the first systematic overview of the efficient training of Transformers, covering the recent progress in acceleration arithmetic and hardware, with a focus on the former. We analyze and compare methods that save computation and memory costs for intermediate tensors during training, t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.01107","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.01107/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.01107","created_at":"2026-07-05T06:06:57.688673+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.01107v3","created_at":"2026-07-05T06:06:57.688673+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.01107","created_at":"2026-07-05T06:06:57.688673+00:00"},{"alias_kind":"pith_short_12","alias_value":"PDXHAZK6X6HR","created_at":"2026-07-05T06:06:57.688673+00:00"},{"alias_kind":"pith_short_16","alias_value":"PDXHAZK6X6HRW5C6","created_at":"2026-07-05T06:06:57.688673+00:00"},{"alias_kind":"pith_short_8","alias_value":"PDXHAZK6","created_at":"2026-07-05T06:06:57.688673+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06930","citing_title":"Imputation Meets Clustering: Exploiting Latent Subgroup Structure for Missing Data Recovery","ref_index":50,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19797","citing_title":"Improving End-to-End Speech Recognition for Dysarthric Speech through In-Domain Data Augmentation","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23292","citing_title":"Agentic Physical AI toward a Domain-Specific Foundation Model for Nuclear Reactor Control","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09176","citing_title":"Navigating LLM Valley: From AdamW to Memory-Efficient and Matrix-Based Optimizers","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PDXHAZK6X6HRW5C6ZWPFNAQCJY","json":"https://pith.science/pith/PDXHAZK6X6HRW5C6ZWPFNAQCJY.json","graph_json":"https://pith.science/api/pith-number/PDXHAZK6X6HRW5C6ZWPFNAQCJY/graph.json","events_json":"https://pith.science/api/pith-number/PDXHAZK6X6HRW5C6ZWPFNAQCJY/events.json","paper":"https://pith.science/paper/PDXHAZK6"},"agent_actions":{"view_html":"https://pith.science/pith/PDXHAZK6X6HRW5C6ZWPFNAQCJY","download_json":"https://pith.science/pith/PDXHAZK6X6HRW5C6ZWPFNAQCJY.json","view_paper":"https://pith.science/paper/PDXHAZK6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.01107&json=true","fetch_graph":"https://pith.science/api/pith-number/PDXHAZK6X6HRW5C6ZWPFNAQCJY/graph.json","fetch_events":"https://pith.science/api/pith-number/PDXHAZK6X6HRW5C6ZWPFNAQCJY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PDXHAZK6X6HRW5C6ZWPFNAQCJY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PDXHAZK6X6HRW5C6ZWPFNAQCJY/action/storage_attestation","attest_author":"https://pith.science/pith/PDXHAZK6X6HRW5C6ZWPFNAQCJY/action/author_attestation","sign_citation":"https://pith.science/pith/PDXHAZK6X6HRW5C6ZWPFNAQCJY/action/citation_signature","submit_replication":"https://pith.science/pith/PDXHAZK6X6HRW5C6ZWPFNAQCJY/action/replication_record"}},"created_at":"2026-07-05T06:06:57.688673+00:00","updated_at":"2026-07-05T06:06:57.688673+00:00"}