{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:4PVEPCN5QIS3TVBXZWVZLEAFNV","short_pith_number":"pith:4PVEPCN5","schema_version":"1.0","canonical_sha256":"e3ea4789bd8225b9d437cdab9590056d6ac937fe94fd42859fe1c0f9dbf153c6","source":{"kind":"arxiv","id":"2109.08668","version":2},"attestation_state":"computed","paper":{"title":"Primer: Searching for Efficient Transformers for Language Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.NE"],"primary_cat":"cs.LG","authors_text":"David R. So, Hanxiao Liu, Noam Shazeer, Quoc V. Le, Wojciech Ma\\'nke, Zihang Dai","submitted_at":"2021-09-17T17:50:39Z","abstract_excerpt":"Large Transformer models have been central to recent advances in natural language processing. The training and inference costs of these models, however, have grown rapidly and become prohibitively expensive. Here we aim to reduce the costs of Transformers by searching for a more efficient variant. Compared to previous approaches, our search is performed at a lower level, over the primitives that define a Transformer TensorFlow program. We identify an architecture, named Primer, that has a smaller training cost than the original Transformer and other variants for auto-regressive language modeli"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.08668","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-09-17T17:50:39Z","cross_cats_sorted":["cs.AI","cs.CL","cs.NE"],"title_canon_sha256":"adf5d6af2c24c4d193cc82391b0fa21bd1ed423bc104ec2b61f7f3194b057e85","abstract_canon_sha256":"bde3083a991521284c24f46648bdaed516a5065792d036691ffd78af20ae5773"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:51:09.771192Z","signature_b64":"98Mk5rPyhynUsok5Z1x3UYWEUZw2LQhsquVxtZ5snLj2f05wDy5/HZo6RbcYzqZnaBhS9eN79Pn78WTkTXU2Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e3ea4789bd8225b9d437cdab9590056d6ac937fe94fd42859fe1c0f9dbf153c6","last_reissued_at":"2026-07-05T03:51:09.770729Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:51:09.770729Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Primer: Searching for Efficient Transformers for Language Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.NE"],"primary_cat":"cs.LG","authors_text":"David R. So, Hanxiao Liu, Noam Shazeer, Quoc V. Le, Wojciech Ma\\'nke, Zihang Dai","submitted_at":"2021-09-17T17:50:39Z","abstract_excerpt":"Large Transformer models have been central to recent advances in natural language processing. The training and inference costs of these models, however, have grown rapidly and become prohibitively expensive. Here we aim to reduce the costs of Transformers by searching for a more efficient variant. Compared to previous approaches, our search is performed at a lower level, over the primitives that define a Transformer TensorFlow program. We identify an architecture, named Primer, that has a smaller training cost than the original Transformer and other variants for auto-regressive language modeli"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.08668","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.08668/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.08668","created_at":"2026-07-05T03:51:09.770787+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.08668v2","created_at":"2026-07-05T03:51:09.770787+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.08668","created_at":"2026-07-05T03:51:09.770787+00:00"},{"alias_kind":"pith_short_12","alias_value":"4PVEPCN5QIS3","created_at":"2026-07-05T03:51:09.770787+00:00"},{"alias_kind":"pith_short_16","alias_value":"4PVEPCN5QIS3TVBX","created_at":"2026-07-05T03:51:09.770787+00:00"},{"alias_kind":"pith_short_8","alias_value":"4PVEPCN5","created_at":"2026-07-05T03:51:09.770787+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25966","citing_title":"Mapping the Schedule x Bit-Width Boundary in Sub-100M Quantisation-Aware Training","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12744","citing_title":"Resting Neurons, Active Insights: Robustifying Activation Sparsity in LLMs via Spontaneity","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17659","citing_title":"Bug or Feature$^2$: Weight Drift, Activation Sparsity and Spikes","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17659","citing_title":"Bug or Feature$^2$: Weight Drift, Activation Sparsity and Spikes","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2512.20856","citing_title":"NVIDIA Nemotron 3: Efficient and Open Intelligence","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2211.17192","citing_title":"Fast Inference from Transformers via Speculative Decoding","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12744","citing_title":"Resting Neurons, Active Insights: Robustifying Activation Sparsity in LLMs via Spontaneity","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2602.04572","citing_title":"From Competition to Collaboration: Designing Sustainable Mechanisms Between LLMs and Online Forums","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2202.08906","citing_title":"ST-MoE: Designing Stable and Transferable Sparse Expert Models","ref_index":200,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10775","citing_title":"On the global convergence of gradient descent for wide shallow models with bounded nonlinearities","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2204.14198","citing_title":"Flamingo: a Visual Language Model for Few-Shot Learning","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03667","citing_title":"ELAS: Efficient Pre-Training of Low-Rank Large Language Models via 2:4 Activation Sparsity","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14430","citing_title":"Three-Phase Transformer","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4PVEPCN5QIS3TVBXZWVZLEAFNV","json":"https://pith.science/pith/4PVEPCN5QIS3TVBXZWVZLEAFNV.json","graph_json":"https://pith.science/api/pith-number/4PVEPCN5QIS3TVBXZWVZLEAFNV/graph.json","events_json":"https://pith.science/api/pith-number/4PVEPCN5QIS3TVBXZWVZLEAFNV/events.json","paper":"https://pith.science/paper/4PVEPCN5"},"agent_actions":{"view_html":"https://pith.science/pith/4PVEPCN5QIS3TVBXZWVZLEAFNV","download_json":"https://pith.science/pith/4PVEPCN5QIS3TVBXZWVZLEAFNV.json","view_paper":"https://pith.science/paper/4PVEPCN5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.08668&json=true","fetch_graph":"https://pith.science/api/pith-number/4PVEPCN5QIS3TVBXZWVZLEAFNV/graph.json","fetch_events":"https://pith.science/api/pith-number/4PVEPCN5QIS3TVBXZWVZLEAFNV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4PVEPCN5QIS3TVBXZWVZLEAFNV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4PVEPCN5QIS3TVBXZWVZLEAFNV/action/storage_attestation","attest_author":"https://pith.science/pith/4PVEPCN5QIS3TVBXZWVZLEAFNV/action/author_attestation","sign_citation":"https://pith.science/pith/4PVEPCN5QIS3TVBXZWVZLEAFNV/action/citation_signature","submit_replication":"https://pith.science/pith/4PVEPCN5QIS3TVBXZWVZLEAFNV/action/replication_record"}},"created_at":"2026-07-05T03:51:09.770787+00:00","updated_at":"2026-07-05T03:51:09.770787+00:00"}