{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:4VVNV6AWKD3S633LJ36KBYNBTN","short_pith_number":"pith:4VVNV6AW","schema_version":"1.0","canonical_sha256":"e56adaf81650f72f6f6b4efca0e1a19b5760823496de78e8b49df18afd51fff8","source":{"kind":"arxiv","id":"2106.02242","version":2},"attestation_state":"computed","paper":{"title":"Scalable Transformers for Neural Machine Translation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hongsheng Li, Jifeng Dai, Peng Gao, Shijie Geng, Xiaogang Wang, Yu Qiao","submitted_at":"2021-06-04T04:04:10Z","abstract_excerpt":"Transformer has been widely adopted in Neural Machine Translation (NMT) because of its large capacity and parallel training of sequence generation. However, the deployment of Transformer is challenging because different scenarios require models of different complexities and scales. Naively training multiple Transformers is redundant in terms of both computation and memory. In this paper, we propose a novel Scalable Transformers, which naturally contains sub-Transformers of different scales and have shared parameters. Each sub-Transformer can be easily obtained by cropping the parameters of the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.02242","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-06-04T04:04:10Z","cross_cats_sorted":[],"title_canon_sha256":"1f23793fc8d79a9bbf64de1aaa466a6f5a977d49d10c4202d51f12e472464a0c","abstract_canon_sha256":"9231968146a38cedc4a81e33deb93b25553c4b7d34023b596c4b547846f7471d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:50:30.547729Z","signature_b64":"nQ7zvk/zdYgIvbmfyva2cklpW7NrqjLkLbN5TIjvVbujkzsEu/mr/fDimWLW7LpAC8FEgAr27o96Yc8yw77bBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e56adaf81650f72f6f6b4efca0e1a19b5760823496de78e8b49df18afd51fff8","last_reissued_at":"2026-07-05T02:50:30.547343Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:50:30.547343Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scalable Transformers for Neural Machine Translation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hongsheng Li, Jifeng Dai, Peng Gao, Shijie Geng, Xiaogang Wang, Yu Qiao","submitted_at":"2021-06-04T04:04:10Z","abstract_excerpt":"Transformer has been widely adopted in Neural Machine Translation (NMT) because of its large capacity and parallel training of sequence generation. However, the deployment of Transformer is challenging because different scenarios require models of different complexities and scales. Naively training multiple Transformers is redundant in terms of both computation and memory. In this paper, we propose a novel Scalable Transformers, which naturally contains sub-Transformers of different scales and have shared parameters. Each sub-Transformer can be easily obtained by cropping the parameters of the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.02242","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.02242/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.02242","created_at":"2026-07-05T02:50:30.547400+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.02242v2","created_at":"2026-07-05T02:50:30.547400+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.02242","created_at":"2026-07-05T02:50:30.547400+00:00"},{"alias_kind":"pith_short_12","alias_value":"4VVNV6AWKD3S","created_at":"2026-07-05T02:50:30.547400+00:00"},{"alias_kind":"pith_short_16","alias_value":"4VVNV6AWKD3S633L","created_at":"2026-07-05T02:50:30.547400+00:00"},{"alias_kind":"pith_short_8","alias_value":"4VVNV6AW","created_at":"2026-07-05T02:50:30.547400+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4VVNV6AWKD3S633LJ36KBYNBTN","json":"https://pith.science/pith/4VVNV6AWKD3S633LJ36KBYNBTN.json","graph_json":"https://pith.science/api/pith-number/4VVNV6AWKD3S633LJ36KBYNBTN/graph.json","events_json":"https://pith.science/api/pith-number/4VVNV6AWKD3S633LJ36KBYNBTN/events.json","paper":"https://pith.science/paper/4VVNV6AW"},"agent_actions":{"view_html":"https://pith.science/pith/4VVNV6AWKD3S633LJ36KBYNBTN","download_json":"https://pith.science/pith/4VVNV6AWKD3S633LJ36KBYNBTN.json","view_paper":"https://pith.science/paper/4VVNV6AW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.02242&json=true","fetch_graph":"https://pith.science/api/pith-number/4VVNV6AWKD3S633LJ36KBYNBTN/graph.json","fetch_events":"https://pith.science/api/pith-number/4VVNV6AWKD3S633LJ36KBYNBTN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4VVNV6AWKD3S633LJ36KBYNBTN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4VVNV6AWKD3S633LJ36KBYNBTN/action/storage_attestation","attest_author":"https://pith.science/pith/4VVNV6AWKD3S633LJ36KBYNBTN/action/author_attestation","sign_citation":"https://pith.science/pith/4VVNV6AWKD3S633LJ36KBYNBTN/action/citation_signature","submit_replication":"https://pith.science/pith/4VVNV6AWKD3S633LJ36KBYNBTN/action/replication_record"}},"created_at":"2026-07-05T02:50:30.547400+00:00","updated_at":"2026-07-05T02:50:30.547400+00:00"}