{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:F3QWMBDMZDLCKEVZFSRYRWIWD3","short_pith_number":"pith:F3QWMBDM","schema_version":"1.0","canonical_sha256":"2ee166046cc8d62512b92ca388d9161ed29b6741801a1259808e4f7e43bea656","source":{"kind":"arxiv","id":"2002.11794","version":2},"attestation_state":"computed","paper":{"title":"Train Large, Then Compress: Rethinking Model Size for Efficient Training and Inference of Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Dan Klein, Eric Wallace, Joseph E. Gonzalez, Kevin Lin, Kurt Keutzer, Sheng Shen, Zhuohan Li","submitted_at":"2020-02-26T21:17:13Z","abstract_excerpt":"Since hardware resources are limited, the objective of training deep learning models is typically to maximize accuracy subject to the time and memory constraints of training and inference. We study the impact of model size in this setting, focusing on Transformer models for NLP tasks that are limited by compute: self-supervised pretraining and high-resource machine translation. We first show that even though smaller Transformer models execute faster per iteration, wider and deeper models converge in significantly fewer steps. Moreover, this acceleration in convergence typically outpaces the ad"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.11794","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-02-26T21:17:13Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"82a457c46b8be5ec81937b70973f6cbc8303a4422f3e5ab0fd0b20726065af62","abstract_canon_sha256":"8a516ac4e6042e3549e162c1ef7ee900ef66e1c349bacf27fff284d6636d5b71"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:12:02.724066Z","signature_b64":"LQMCHDclH+jNYqaVsrY1+d1P4xmc/Ds+15eNykV8JQTtLgSOv1NEPiPqH+NYvUNm+tR4WumGm2is9plgOeJrDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2ee166046cc8d62512b92ca388d9161ed29b6741801a1259808e4f7e43bea656","last_reissued_at":"2026-07-05T01:12:02.723674Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:12:02.723674Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Train Large, Then Compress: Rethinking Model Size for Efficient Training and Inference of Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Dan Klein, Eric Wallace, Joseph E. Gonzalez, Kevin Lin, Kurt Keutzer, Sheng Shen, Zhuohan Li","submitted_at":"2020-02-26T21:17:13Z","abstract_excerpt":"Since hardware resources are limited, the objective of training deep learning models is typically to maximize accuracy subject to the time and memory constraints of training and inference. We study the impact of model size in this setting, focusing on Transformer models for NLP tasks that are limited by compute: self-supervised pretraining and high-resource machine translation. We first show that even though smaller Transformer models execute faster per iteration, wider and deeper models converge in significantly fewer steps. Moreover, this acceleration in convergence typically outpaces the ad"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.11794","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.11794/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.11794","created_at":"2026-07-05T01:12:02.723731+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.11794v2","created_at":"2026-07-05T01:12:02.723731+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.11794","created_at":"2026-07-05T01:12:02.723731+00:00"},{"alias_kind":"pith_short_12","alias_value":"F3QWMBDMZDLC","created_at":"2026-07-05T01:12:02.723731+00:00"},{"alias_kind":"pith_short_16","alias_value":"F3QWMBDMZDLCKEVZ","created_at":"2026-07-05T01:12:02.723731+00:00"},{"alias_kind":"pith_short_8","alias_value":"F3QWMBDM","created_at":"2026-07-05T01:12:02.723731+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2102.01293","citing_title":"Scaling Laws for Transfer","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03110","citing_title":"Multi-Aspect Knowledge Distillation for Language Model with Low-rank Factorization","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2010.14701","citing_title":"Scaling Laws for Autoregressive Generative Modeling","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2112.00861","citing_title":"A General Language Assistant as a Laboratory for Alignment","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2207.05221","citing_title":"Language Models (Mostly) Know What They Know","ref_index":129,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F3QWMBDMZDLCKEVZFSRYRWIWD3","json":"https://pith.science/pith/F3QWMBDMZDLCKEVZFSRYRWIWD3.json","graph_json":"https://pith.science/api/pith-number/F3QWMBDMZDLCKEVZFSRYRWIWD3/graph.json","events_json":"https://pith.science/api/pith-number/F3QWMBDMZDLCKEVZFSRYRWIWD3/events.json","paper":"https://pith.science/paper/F3QWMBDM"},"agent_actions":{"view_html":"https://pith.science/pith/F3QWMBDMZDLCKEVZFSRYRWIWD3","download_json":"https://pith.science/pith/F3QWMBDMZDLCKEVZFSRYRWIWD3.json","view_paper":"https://pith.science/paper/F3QWMBDM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.11794&json=true","fetch_graph":"https://pith.science/api/pith-number/F3QWMBDMZDLCKEVZFSRYRWIWD3/graph.json","fetch_events":"https://pith.science/api/pith-number/F3QWMBDMZDLCKEVZFSRYRWIWD3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F3QWMBDMZDLCKEVZFSRYRWIWD3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F3QWMBDMZDLCKEVZFSRYRWIWD3/action/storage_attestation","attest_author":"https://pith.science/pith/F3QWMBDMZDLCKEVZFSRYRWIWD3/action/author_attestation","sign_citation":"https://pith.science/pith/F3QWMBDMZDLCKEVZFSRYRWIWD3/action/citation_signature","submit_replication":"https://pith.science/pith/F3QWMBDMZDLCKEVZFSRYRWIWD3/action/replication_record"}},"created_at":"2026-07-05T01:12:02.723731+00:00","updated_at":"2026-07-05T01:12:02.723731+00:00"}