{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DMW235KFMIHWRCSCBOO7UEE5ZS","short_pith_number":"pith:DMW235KF","schema_version":"1.0","canonical_sha256":"1b2dadf545620f688a420b9dfa109dcc8106498f4e83b395f8f644dfa076cfa7","source":{"kind":"arxiv","id":"2505.23735","version":1},"attestation_state":"computed","paper":{"title":"ATLAS: Learning to Optimally Memorize the Context at Test Time","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ali Behrouz, Majid Daliri, Meisam Razaviyayn, Peilin Zhong, Praneeth Kacham, Vahab Mirrokni, Yuan Deng, Zeman Li","submitted_at":"2025-05-29T17:57:16Z","abstract_excerpt":"Transformers have been established as the most popular backbones in sequence modeling, mainly due to their effectiveness in in-context retrieval tasks and the ability to learn at scale. Their quadratic memory and time complexity, however, bound their applicability in longer sequences and so has motivated researchers to explore effective alternative architectures such as modern recurrent neural networks (a.k.a long-term recurrent memory module). Despite their recent success in diverse downstream tasks, they struggle in tasks that requires long context understanding and extrapolation to longer s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23735","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-29T17:57:16Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"cff327fda3dbb199e904f986ae843be80562b202bf10e21dd543ec694e914222","abstract_canon_sha256":"556b1e8caeb19c25aef37a128bd52bbb08b733c7562e123c6e26bedc632a5864"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:13.454324Z","signature_b64":"gi/4NTii5QfBtyYqypSyN1gpW41ebEp5uCIGdawS73CvVO2TGvR79hdozTV+JAKekp2bZmTGLepvunlxuuKeCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1b2dadf545620f688a420b9dfa109dcc8106498f4e83b395f8f644dfa076cfa7","last_reissued_at":"2026-07-05T11:12:13.453847Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:13.453847Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ATLAS: Learning to Optimally Memorize the Context at Test Time","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ali Behrouz, Majid Daliri, Meisam Razaviyayn, Peilin Zhong, Praneeth Kacham, Vahab Mirrokni, Yuan Deng, Zeman Li","submitted_at":"2025-05-29T17:57:16Z","abstract_excerpt":"Transformers have been established as the most popular backbones in sequence modeling, mainly due to their effectiveness in in-context retrieval tasks and the ability to learn at scale. Their quadratic memory and time complexity, however, bound their applicability in longer sequences and so has motivated researchers to explore effective alternative architectures such as modern recurrent neural networks (a.k.a long-term recurrent memory module). Despite their recent success in diverse downstream tasks, they struggle in tasks that requires long context understanding and extrapolation to longer s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23735","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23735/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23735","created_at":"2026-07-05T11:12:13.453900+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23735v1","created_at":"2026-07-05T11:12:13.453900+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23735","created_at":"2026-07-05T11:12:13.453900+00:00"},{"alias_kind":"pith_short_12","alias_value":"DMW235KFMIHW","created_at":"2026-07-05T11:12:13.453900+00:00"},{"alias_kind":"pith_short_16","alias_value":"DMW235KFMIHWRCSC","created_at":"2026-07-05T11:12:13.453900+00:00"},{"alias_kind":"pith_short_8","alias_value":"DMW235KF","created_at":"2026-07-05T11:12:13.453900+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25342","citing_title":"Lifelong In-Context Learning with Transformers Requires Parametric Forms of Attention","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23670","citing_title":"Tapered Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00368","citing_title":"Beyond Perplexity: A Behavioral Evaluation Framework for Deployment-Memory Claims in LLM Test-Time Training","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02775","citing_title":"AURA: Action-Gated Memory for Robot Policies at Constant VRAM","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31163","citing_title":"Memory by Design: Probabilistic Sequence Layers","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02772","citing_title":"Linearizing Vision Transformer with Test-Time Training","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2510.27258","citing_title":"Higher-order Linear Attention","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2511.07328","citing_title":"Q-RAG: Long Context Multi-step Retrieval via Value-based Embedder Training","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2512.01643","citing_title":"ViT$^3$: Unlocking Test-Time Training in Vision","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2602.21204","citing_title":"Test-Time Training with KV Binding Is Secretly Linear Attention","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13473","citing_title":"OSDN: Improving Delta Rule with Provable Online Preconditioning in Linear Attention","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26692","citing_title":"Kimi Linear: An Expressive, Efficient Attention Architecture","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05838","citing_title":"MDN: Parallelizing Stepwise Momentum for Delta Linear Attention","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16484","citing_title":"DexWorldModel: Causal Latent World Modeling towards Automated Learning of Embodied Tasks","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07350","citing_title":"Fast Spatial Memory with Elastic Test-Time Training","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07279","citing_title":"Mem3R: Streaming 3D Reconstruction with Hybrid Memory via Test-Time Training","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18131","citing_title":"Training LLM Agents for Spontaneous, Reward-Free Self-Evolution via World Knowledge Exploration","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21100","citing_title":"Preconditioned DeltaNet: Curvature-aware Sequence Modeling for Linear Recurrences","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02772","citing_title":"Linearizing Vision Transformer with Test-Time Training","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DMW235KFMIHWRCSCBOO7UEE5ZS","json":"https://pith.science/pith/DMW235KFMIHWRCSCBOO7UEE5ZS.json","graph_json":"https://pith.science/api/pith-number/DMW235KFMIHWRCSCBOO7UEE5ZS/graph.json","events_json":"https://pith.science/api/pith-number/DMW235KFMIHWRCSCBOO7UEE5ZS/events.json","paper":"https://pith.science/paper/DMW235KF"},"agent_actions":{"view_html":"https://pith.science/pith/DMW235KFMIHWRCSCBOO7UEE5ZS","download_json":"https://pith.science/pith/DMW235KFMIHWRCSCBOO7UEE5ZS.json","view_paper":"https://pith.science/paper/DMW235KF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23735&json=true","fetch_graph":"https://pith.science/api/pith-number/DMW235KFMIHWRCSCBOO7UEE5ZS/graph.json","fetch_events":"https://pith.science/api/pith-number/DMW235KFMIHWRCSCBOO7UEE5ZS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DMW235KFMIHWRCSCBOO7UEE5ZS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DMW235KFMIHWRCSCBOO7UEE5ZS/action/storage_attestation","attest_author":"https://pith.science/pith/DMW235KFMIHWRCSCBOO7UEE5ZS/action/author_attestation","sign_citation":"https://pith.science/pith/DMW235KFMIHWRCSCBOO7UEE5ZS/action/citation_signature","submit_replication":"https://pith.science/pith/DMW235KFMIHWRCSCBOO7UEE5ZS/action/replication_record"}},"created_at":"2026-07-05T11:12:13.453900+00:00","updated_at":"2026-07-05T11:12:13.453900+00:00"}