{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JBOMINV7QLJKP7IQGFRT33JU6W","short_pith_number":"pith:JBOMINV7","schema_version":"1.0","canonical_sha256":"485cc436bf82d2a7fd1031633ded34f59ae0b41f6283296145352d8ff624fa2f","source":{"kind":"arxiv","id":"2407.05483","version":1},"attestation_state":"computed","paper":{"title":"Just read twice: closing the recall gap for recurrent language models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aaryan Singhal, Aman Timalsina, Ashish Rao, Atri Rudra, Benjamin Spector, Christopher R\\'e, Sabri Eyuboglu, Simran Arora, Xinyi Zhao","submitted_at":"2024-07-07T19:55:09Z","abstract_excerpt":"Recurrent large language models that compete with Transformers in language modeling perplexity are emerging at a rapid rate (e.g., Mamba, RWKV). Excitingly, these architectures use a constant amount of memory during inference. However, due to the limited memory, recurrent LMs cannot recall and use all the information in long contexts leading to brittle in-context learning (ICL) quality. A key challenge for efficient LMs is selecting what information to store versus discard. In this work, we observe the order in which information is shown to the LM impacts the selection difficulty. To formalize"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.05483","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-07T19:55:09Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"9255a364f44c19fc18058289255b873e2c6de7fa1d409ac34ccb87e423d1f4b0","abstract_canon_sha256":"0ece68899cccaacbebb7882515dccf9419e9c06d5d7db213789ad340defb0939"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:41:14.103345Z","signature_b64":"kICXu3DIOzPegai+JCMMY6hASj8YlCQmLrUmWd92obohzHYk/jGqBqw3P0l+xixvVTkHk6h0HPUN0vv4Ie79AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"485cc436bf82d2a7fd1031633ded34f59ae0b41f6283296145352d8ff624fa2f","last_reissued_at":"2026-07-05T08:41:14.102892Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:41:14.102892Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Just read twice: closing the recall gap for recurrent language models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aaryan Singhal, Aman Timalsina, Ashish Rao, Atri Rudra, Benjamin Spector, Christopher R\\'e, Sabri Eyuboglu, Simran Arora, Xinyi Zhao","submitted_at":"2024-07-07T19:55:09Z","abstract_excerpt":"Recurrent large language models that compete with Transformers in language modeling perplexity are emerging at a rapid rate (e.g., Mamba, RWKV). Excitingly, these architectures use a constant amount of memory during inference. However, due to the limited memory, recurrent LMs cannot recall and use all the information in long contexts leading to brittle in-context learning (ICL) quality. A key challenge for efficient LMs is selecting what information to store versus discard. In this work, we observe the order in which information is shown to the LM impacts the selection difficulty. To formalize"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.05483","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.05483/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.05483","created_at":"2026-07-05T08:41:14.102960+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.05483v1","created_at":"2026-07-05T08:41:14.102960+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.05483","created_at":"2026-07-05T08:41:14.102960+00:00"},{"alias_kind":"pith_short_12","alias_value":"JBOMINV7QLJK","created_at":"2026-07-05T08:41:14.102960+00:00"},{"alias_kind":"pith_short_16","alias_value":"JBOMINV7QLJKP7IQ","created_at":"2026-07-05T08:41:14.102960+00:00"},{"alias_kind":"pith_short_8","alias_value":"JBOMINV7","created_at":"2026-07-05T08:41:14.102960+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25342","citing_title":"Lifelong In-Context Learning with Transformers Requires Parametric Forms of Attention","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01792","citing_title":"PARTREP: Learning What to Repeat for Decoder-only LLMs","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22630","citing_title":"StateX: Enhancing RNN Recall via Post-training State Expansion","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13473","citing_title":"OSDN: Improving Delta Rule with Provable Online Preconditioning in Linear Attention","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2412.06464","citing_title":"Gated Delta Networks: Improving Mamba2 with Delta Rule","ref_index":297,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03953","citing_title":"Transformers with Selective Access to Early Representations","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03953","citing_title":"Transformers with Selective Access to Early Representations","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JBOMINV7QLJKP7IQGFRT33JU6W","json":"https://pith.science/pith/JBOMINV7QLJKP7IQGFRT33JU6W.json","graph_json":"https://pith.science/api/pith-number/JBOMINV7QLJKP7IQGFRT33JU6W/graph.json","events_json":"https://pith.science/api/pith-number/JBOMINV7QLJKP7IQGFRT33JU6W/events.json","paper":"https://pith.science/paper/JBOMINV7"},"agent_actions":{"view_html":"https://pith.science/pith/JBOMINV7QLJKP7IQGFRT33JU6W","download_json":"https://pith.science/pith/JBOMINV7QLJKP7IQGFRT33JU6W.json","view_paper":"https://pith.science/paper/JBOMINV7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.05483&json=true","fetch_graph":"https://pith.science/api/pith-number/JBOMINV7QLJKP7IQGFRT33JU6W/graph.json","fetch_events":"https://pith.science/api/pith-number/JBOMINV7QLJKP7IQGFRT33JU6W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JBOMINV7QLJKP7IQGFRT33JU6W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JBOMINV7QLJKP7IQGFRT33JU6W/action/storage_attestation","attest_author":"https://pith.science/pith/JBOMINV7QLJKP7IQGFRT33JU6W/action/author_attestation","sign_citation":"https://pith.science/pith/JBOMINV7QLJKP7IQGFRT33JU6W/action/citation_signature","submit_replication":"https://pith.science/pith/JBOMINV7QLJKP7IQGFRT33JU6W/action/replication_record"}},"created_at":"2026-07-05T08:41:14.102960+00:00","updated_at":"2026-07-05T08:41:14.102960+00:00"}