{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BL7MJCYOZOJURKLRIBVJVJT6CC","short_pith_number":"pith:BL7MJCYO","schema_version":"1.0","canonical_sha256":"0afec48b0ecb9348a971406a9aa67e10a591356c1d8360485fe88500dd5b1de5","source":{"kind":"arxiv","id":"2305.16300","version":2},"attestation_state":"computed","paper":{"title":"Landmark Attention: Random-Access Infinite Context Length for Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Amirkeivan Mohtashami, Martin Jaggi","submitted_at":"2023-05-25T17:53:42Z","abstract_excerpt":"While Transformers have shown remarkable success in natural language processing, their attention mechanism's large memory requirements have limited their ability to handle longer contexts. Prior approaches, such as recurrent memory or retrieval-based augmentation, have either compromised the random-access flexibility of attention (i.e., the capability to select any token in the entire context) or relied on separate mechanisms for relevant context retrieval, which may not be compatible with the model's attention. In this paper, we present a novel approach that allows access to the complete cont"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.16300","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-05-25T17:53:42Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"133ab8c820b0995e9c47d324fc3e935756266da02ea1f9faba39712c5fbf3096","abstract_canon_sha256":"24fe12071f57030322800927bd7e5e5d756c7e8815a7211b8fec86667c1b596a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:14:12.680554Z","signature_b64":"RLb50flzTQauQNvkW9EkqIx0tk7+HnkWZ39Epqj6JgRUyVjT9JkEm/erxhvbU9FyxOK/Rvc/KWde7RwDr4qhAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0afec48b0ecb9348a971406a9aa67e10a591356c1d8360485fe88500dd5b1de5","last_reissued_at":"2026-07-05T07:14:12.680034Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:14:12.680034Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Landmark Attention: Random-Access Infinite Context Length for Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Amirkeivan Mohtashami, Martin Jaggi","submitted_at":"2023-05-25T17:53:42Z","abstract_excerpt":"While Transformers have shown remarkable success in natural language processing, their attention mechanism's large memory requirements have limited their ability to handle longer contexts. Prior approaches, such as recurrent memory or retrieval-based augmentation, have either compromised the random-access flexibility of attention (i.e., the capability to select any token in the entire context) or relied on separate mechanisms for relevant context retrieval, which may not be compatible with the model's attention. In this paper, we present a novel approach that allows access to the complete cont"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.16300","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.16300/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.16300","created_at":"2026-07-05T07:14:12.680092+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.16300v2","created_at":"2026-07-05T07:14:12.680092+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.16300","created_at":"2026-07-05T07:14:12.680092+00:00"},{"alias_kind":"pith_short_12","alias_value":"BL7MJCYOZOJU","created_at":"2026-07-05T07:14:12.680092+00:00"},{"alias_kind":"pith_short_16","alias_value":"BL7MJCYOZOJURKLR","created_at":"2026-07-05T07:14:12.680092+00:00"},{"alias_kind":"pith_short_8","alias_value":"BL7MJCYO","created_at":"2026-07-05T07:14:12.680092+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":83,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":191,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02303","citing_title":"A Hippocampus for Linear Attention: An Exact Memory for What the Recurrent State Forgets","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26797","citing_title":"Latent Recurrent Transformer: Architecture Exploration, Training Strategies, and Scaling Behavior","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2401.04088","citing_title":"Mixtral of Experts","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2502.01941","citing_title":"Semantic Integrity Matters: Benchmarking and Preserving High-Density Reasoning in KV Cache Compression","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21016","citing_title":"Gated KalmaNet: A Fading Memory Layer Through Test-Time Ridge Regression","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17625","citing_title":"Episodic-Semantic Memory Architecture for Long-Horizon Scientific Agents","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2601.04237","citing_title":"SAGE-32B: Agentic Reasoning via Iterative Distillation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2402.13753","citing_title":"LongRoPE: Extending LLM Context Window Beyond 2 Million Tokens","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2306.15595","citing_title":"Extending Context Window of Large Language Models via Positional Interpolation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2308.14508","citing_title":"LongBench: A Bilingual, Multitask Benchmark for Long Context Understanding","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2402.02750","citing_title":"KIVI: A Tuning-Free Asymmetric 2bit Quantization for KV Cache","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2309.00071","citing_title":"YaRN: Efficient Context Window Extension of Large Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2309.07864","citing_title":"The Rise and Potential of Large Language Model Based Agents: A Survey","ref_index":234,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10539","citing_title":"IceCache: Memory-efficient KV-cache Management for Long-Sequence LLMs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12913","citing_title":"CoDe-R: Refining Decompiler Output with LLMs via Rationale Guidance and Adaptive Inference","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BL7MJCYOZOJURKLRIBVJVJT6CC","json":"https://pith.science/pith/BL7MJCYOZOJURKLRIBVJVJT6CC.json","graph_json":"https://pith.science/api/pith-number/BL7MJCYOZOJURKLRIBVJVJT6CC/graph.json","events_json":"https://pith.science/api/pith-number/BL7MJCYOZOJURKLRIBVJVJT6CC/events.json","paper":"https://pith.science/paper/BL7MJCYO"},"agent_actions":{"view_html":"https://pith.science/pith/BL7MJCYOZOJURKLRIBVJVJT6CC","download_json":"https://pith.science/pith/BL7MJCYOZOJURKLRIBVJVJT6CC.json","view_paper":"https://pith.science/paper/BL7MJCYO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.16300&json=true","fetch_graph":"https://pith.science/api/pith-number/BL7MJCYOZOJURKLRIBVJVJT6CC/graph.json","fetch_events":"https://pith.science/api/pith-number/BL7MJCYOZOJURKLRIBVJVJT6CC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BL7MJCYOZOJURKLRIBVJVJT6CC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BL7MJCYOZOJURKLRIBVJVJT6CC/action/storage_attestation","attest_author":"https://pith.science/pith/BL7MJCYOZOJURKLRIBVJVJT6CC/action/author_attestation","sign_citation":"https://pith.science/pith/BL7MJCYOZOJURKLRIBVJVJT6CC/action/citation_signature","submit_replication":"https://pith.science/pith/BL7MJCYOZOJURKLRIBVJVJT6CC/action/replication_record"}},"created_at":"2026-07-05T07:14:12.680092+00:00","updated_at":"2026-07-05T07:14:12.680092+00:00"}