{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TAHEU6GAQCK47BPF33Q5RVJLM7","short_pith_number":"pith:TAHEU6GA","schema_version":"1.0","canonical_sha256":"980e4a78c08095cf85e5dee1d8d52b67e9ee2c7027fd06fb8fc784d07788c933","source":{"kind":"arxiv","id":"2304.11062","version":2},"attestation_state":"computed","paper":{"title":"Scaling Transformer to 1M tokens and beyond with RMT","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aydar Bulatov, Mikhail S. Burtsev, Yermek Kapushev, Yuri Kuratov","submitted_at":"2023-04-19T16:18:54Z","abstract_excerpt":"A major limitation for the broader scope of problems solvable by transformers is the quadratic scaling of computational complexity with input size. In this study, we investigate the recurrent memory augmentation of pre-trained transformer models to extend input context length while linearly scaling compute. Our approach demonstrates the capability to store information in memory for sequences of up to an unprecedented two million tokens while maintaining high retrieval accuracy. Experiments with language modeling tasks show perplexity improvement as the number of processed input segments increa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.11062","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-04-19T16:18:54Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"6e644c7b0d1e5ac02561373b22b6e4fda5f99fcd0ab014de40d43355d5ad5ade","abstract_canon_sha256":"f291dc9f8335827de9c74238ecbe466282f515ba163ffd20d41a6a94010e866b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:41:39.407240Z","signature_b64":"+Zx1gUBJp5BAglix+3pA7dafXsiBQp0n4vM4hdI9oob0aYCyxs9+ijQbpHWxDwi4Z7THoNY+ktso7Xpc2cI0AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"980e4a78c08095cf85e5dee1d8d52b67e9ee2c7027fd06fb8fc784d07788c933","last_reissued_at":"2026-07-05T07:41:39.406786Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:41:39.406786Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Transformer to 1M tokens and beyond with RMT","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aydar Bulatov, Mikhail S. Burtsev, Yermek Kapushev, Yuri Kuratov","submitted_at":"2023-04-19T16:18:54Z","abstract_excerpt":"A major limitation for the broader scope of problems solvable by transformers is the quadratic scaling of computational complexity with input size. In this study, we investigate the recurrent memory augmentation of pre-trained transformer models to extend input context length while linearly scaling compute. Our approach demonstrates the capability to store information in memory for sequences of up to an unprecedented two million tokens while maintaining high retrieval accuracy. Experiments with language modeling tasks show perplexity improvement as the number of processed input segments increa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.11062","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.11062/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.11062","created_at":"2026-07-05T07:41:39.406851+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.11062v2","created_at":"2026-07-05T07:41:39.406851+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.11062","created_at":"2026-07-05T07:41:39.406851+00:00"},{"alias_kind":"pith_short_12","alias_value":"TAHEU6GAQCK4","created_at":"2026-07-05T07:41:39.406851+00:00"},{"alias_kind":"pith_short_16","alias_value":"TAHEU6GAQCK47BPF","created_at":"2026-07-05T07:41:39.406851+00:00"},{"alias_kind":"pith_short_8","alias_value":"TAHEU6GA","created_at":"2026-07-05T07:41:39.406851+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01223","citing_title":"Connecting the Dots: Benchmarking Reflective Memory in Long-Horizon Dialogue","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2603.04639","citing_title":"RoboMME: Benchmarking and Understanding Memory for Robotic Generalist Policies","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2309.10668","citing_title":"Language Modeling Is Compression","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2507.02259","citing_title":"MemAgent: Reshaping Long-Context LLM with Multi-Conv RL-based Memory Agent","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00663","citing_title":"Titans: Learning to Memorize at Test Time","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2308.14508","citing_title":"LongBench: A Bilingual, Multitask Benchmark for Long Context Understanding","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18137","citing_title":"AQPIM: Breaking the PIM Capacity Wall for LLMs with In-Memory Activation Quantization","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2404.06654","citing_title":"RULER: What's the Real Context Size of Your Long-Context Language Models?","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2312.00752","citing_title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TAHEU6GAQCK47BPF33Q5RVJLM7","json":"https://pith.science/pith/TAHEU6GAQCK47BPF33Q5RVJLM7.json","graph_json":"https://pith.science/api/pith-number/TAHEU6GAQCK47BPF33Q5RVJLM7/graph.json","events_json":"https://pith.science/api/pith-number/TAHEU6GAQCK47BPF33Q5RVJLM7/events.json","paper":"https://pith.science/paper/TAHEU6GA"},"agent_actions":{"view_html":"https://pith.science/pith/TAHEU6GAQCK47BPF33Q5RVJLM7","download_json":"https://pith.science/pith/TAHEU6GAQCK47BPF33Q5RVJLM7.json","view_paper":"https://pith.science/paper/TAHEU6GA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.11062&json=true","fetch_graph":"https://pith.science/api/pith-number/TAHEU6GAQCK47BPF33Q5RVJLM7/graph.json","fetch_events":"https://pith.science/api/pith-number/TAHEU6GAQCK47BPF33Q5RVJLM7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TAHEU6GAQCK47BPF33Q5RVJLM7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TAHEU6GAQCK47BPF33Q5RVJLM7/action/storage_attestation","attest_author":"https://pith.science/pith/TAHEU6GAQCK47BPF33Q5RVJLM7/action/author_attestation","sign_citation":"https://pith.science/pith/TAHEU6GAQCK47BPF33Q5RVJLM7/action/citation_signature","submit_replication":"https://pith.science/pith/TAHEU6GAQCK47BPF33Q5RVJLM7/action/replication_record"}},"created_at":"2026-07-05T07:41:39.406851+00:00","updated_at":"2026-07-05T07:41:39.406851+00:00"}