{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:CJ47NYSE2T2BQG34PHCL3WM2KK","short_pith_number":"pith:CJ47NYSE","schema_version":"1.0","canonical_sha256":"1279f6e244d4f4181b7c79c4bdd99a52ac8b05323060e620e2bd0f6c49401484","source":{"kind":"arxiv","id":"2207.06366","version":1},"attestation_state":"computed","paper":{"title":"N-Grammer: Augmenting Transformers with latent n-grams","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aurko Roy, Benjamin Lee, Christopher Fifty, Guangda Lai, Jeffrey Zhao, Phuong Dao, Rigel Swavely, Rohan Anil, Shen Wu, Shibo Wang, Shuyuan Zhang, Tao (Alex) Yu, Ye Zhang, Yonghui Wu, Zhifeng Chen","submitted_at":"2022-07-13T17:18:02Z","abstract_excerpt":"Transformer models have recently emerged as one of the foundational models in natural language processing, and as a byproduct, there is significant recent interest and investment in scaling these models. However, the training and inference costs of these large Transformer language models are prohibitive, thus necessitating more research in identifying more efficient variants. In this work, we propose a simple yet effective modification to the Transformer architecture inspired by the literature in statistical language modeling, by augmenting the model with n-grams that are constructed from a di"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2207.06366","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-07-13T17:18:02Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"f31878230ebb8eb4b648d9332c72aedec2622f03226aa5afae9e077f4c30c12f","abstract_canon_sha256":"8b17b398264ea4ccb5932707c7482ca8b3c5e631e915ea04950de182dd1eedd7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:40:04.682894Z","signature_b64":"PacAP9WZUzgC/1CKqyKtA7/3etjaMuYo6aTRF2LFE0m/ZSM1pd83oxZ8jtpyZaFmGCYcvbE0qR7Q7se9qfuJDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1279f6e244d4f4181b7c79c4bdd99a52ac8b05323060e620e2bd0f6c49401484","last_reissued_at":"2026-07-05T04:40:04.682410Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:40:04.682410Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"N-Grammer: Augmenting Transformers with latent n-grams","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aurko Roy, Benjamin Lee, Christopher Fifty, Guangda Lai, Jeffrey Zhao, Phuong Dao, Rigel Swavely, Rohan Anil, Shen Wu, Shibo Wang, Shuyuan Zhang, Tao (Alex) Yu, Ye Zhang, Yonghui Wu, Zhifeng Chen","submitted_at":"2022-07-13T17:18:02Z","abstract_excerpt":"Transformer models have recently emerged as one of the foundational models in natural language processing, and as a byproduct, there is significant recent interest and investment in scaling these models. However, the training and inference costs of these large Transformer language models are prohibitive, thus necessitating more research in identifying more efficient variants. In this work, we propose a simple yet effective modification to the Transformer architecture inspired by the literature in statistical language modeling, by augmenting the model with n-grams that are constructed from a di"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2207.06366","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2207.06366/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2207.06366","created_at":"2026-07-05T04:40:04.682467+00:00"},{"alias_kind":"arxiv_version","alias_value":"2207.06366v1","created_at":"2026-07-05T04:40:04.682467+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2207.06366","created_at":"2026-07-05T04:40:04.682467+00:00"},{"alias_kind":"pith_short_12","alias_value":"CJ47NYSE2T2B","created_at":"2026-07-05T04:40:04.682467+00:00"},{"alias_kind":"pith_short_16","alias_value":"CJ47NYSE2T2BQG34","created_at":"2026-07-05T04:40:04.682467+00:00"},{"alias_kind":"pith_short_8","alias_value":"CJ47NYSE","created_at":"2026-07-05T04:40:04.682467+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.27263","citing_title":"Decoupling the Benefits of Subword Tokenization for Language Model Training via Byte-level Simulation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27263","citing_title":"Decoupling the Benefits of Subword Tokenization for Language Model Training via Byte-level Simulation","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CJ47NYSE2T2BQG34PHCL3WM2KK","json":"https://pith.science/pith/CJ47NYSE2T2BQG34PHCL3WM2KK.json","graph_json":"https://pith.science/api/pith-number/CJ47NYSE2T2BQG34PHCL3WM2KK/graph.json","events_json":"https://pith.science/api/pith-number/CJ47NYSE2T2BQG34PHCL3WM2KK/events.json","paper":"https://pith.science/paper/CJ47NYSE"},"agent_actions":{"view_html":"https://pith.science/pith/CJ47NYSE2T2BQG34PHCL3WM2KK","download_json":"https://pith.science/pith/CJ47NYSE2T2BQG34PHCL3WM2KK.json","view_paper":"https://pith.science/paper/CJ47NYSE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2207.06366&json=true","fetch_graph":"https://pith.science/api/pith-number/CJ47NYSE2T2BQG34PHCL3WM2KK/graph.json","fetch_events":"https://pith.science/api/pith-number/CJ47NYSE2T2BQG34PHCL3WM2KK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CJ47NYSE2T2BQG34PHCL3WM2KK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CJ47NYSE2T2BQG34PHCL3WM2KK/action/storage_attestation","attest_author":"https://pith.science/pith/CJ47NYSE2T2BQG34PHCL3WM2KK/action/author_attestation","sign_citation":"https://pith.science/pith/CJ47NYSE2T2BQG34PHCL3WM2KK/action/citation_signature","submit_replication":"https://pith.science/pith/CJ47NYSE2T2BQG34PHCL3WM2KK/action/replication_record"}},"created_at":"2026-07-05T04:40:04.682467+00:00","updated_at":"2026-07-05T04:40:04.682467+00:00"}