{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EBDWUIG4R5IHRLAHWR2ZNKMJSW","short_pith_number":"pith:EBDWUIG4","schema_version":"1.0","canonical_sha256":"20476a20dc8f5078ac07b47596a98995a3a77ece0bfadc31830e59c6bc813ef4","source":{"kind":"arxiv","id":"2406.06484","version":6},"attestation_state":"computed","paper":{"title":"Parallelizing Linear Transformers with the Delta Rule over Sequence Length","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Bailin Wang, Songlin Yang, Yikang Shen, Yoon Kim, Yu Zhang","submitted_at":"2024-06-10T17:24:42Z","abstract_excerpt":"Transformers with linear attention (i.e., linear transformers) and state-space models have recently been suggested as a viable linear-time alternative to transformers with softmax attention. However, these models still underperform transformers especially on tasks that require in-context retrieval. While more expressive variants of linear transformers which replace the additive update in linear transformers with the delta rule (DeltaNet) have been found to be more effective at associative recall, existing algorithms for training such models do not parallelize over sequence length and are thus "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.06484","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-10T17:24:42Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"d63aa8bed0b7b60494572722cf195fdd23debb45cf26cf0fbc7f609c9d880d11","abstract_canon_sha256":"40a48dc4a6f48e5a06eefda6c1d5479421f6c8cac0ba6fe17df5999ca168bcdc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:01:13.711874Z","signature_b64":"25Mm4xTxM/0vYYqz8OHawC0z2BrnnegPIM2GsxY9r3C9Sfmv80uySEOUkcg0ulg67ly4BQfXIqy5d3MEY9YdDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"20476a20dc8f5078ac07b47596a98995a3a77ece0bfadc31830e59c6bc813ef4","last_reissued_at":"2026-07-05T10:01:13.711378Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:01:13.711378Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Parallelizing Linear Transformers with the Delta Rule over Sequence Length","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Bailin Wang, Songlin Yang, Yikang Shen, Yoon Kim, Yu Zhang","submitted_at":"2024-06-10T17:24:42Z","abstract_excerpt":"Transformers with linear attention (i.e., linear transformers) and state-space models have recently been suggested as a viable linear-time alternative to transformers with softmax attention. However, these models still underperform transformers especially on tasks that require in-context retrieval. While more expressive variants of linear transformers which replace the additive update in linear transformers with the delta rule (DeltaNet) have been found to be more effective at associative recall, existing algorithms for training such models do not parallelize over sequence length and are thus "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.06484","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.06484/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.06484","created_at":"2026-07-05T10:01:13.711439+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.06484v6","created_at":"2026-07-05T10:01:13.711439+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.06484","created_at":"2026-07-05T10:01:13.711439+00:00"},{"alias_kind":"pith_short_12","alias_value":"EBDWUIG4R5IH","created_at":"2026-07-05T10:01:13.711439+00:00"},{"alias_kind":"pith_short_16","alias_value":"EBDWUIG4R5IHRLAH","created_at":"2026-07-05T10:01:13.711439+00:00"},{"alias_kind":"pith_short_8","alias_value":"EBDWUIG4","created_at":"2026-07-05T10:01:13.711439+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":140,"is_internal_anchor":true},{"citing_arxiv_id":"2607.07386","citing_title":"Sparse Delta Memory: Scaling the State of Linear RNNs through Sparsity","ref_index":114,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20474","citing_title":"UltraQuant: 4-bit KV Caching for Context-Heavy Agents","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08804","citing_title":"Q-Delta: Beyond Key-Value Associative State Evolution","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07404","citing_title":"Reversible Foundations: Training a 120B Sparse MoE through State-Preserving Scaling","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06479","citing_title":"Pretraining Recurrent Networks without Recurrence","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21070","citing_title":"Towards Understanding Self-Pretraining for Sequence Classification","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26083","citing_title":"Nirvana: A Specialized Generalist Model With Task-Aware Memory Mechanism","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23884","citing_title":"Test-Time Training Done Right","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2603.04385","citing_title":"ZipMap: Linear-Time Stateful 3D Reconstruction via Test-Time Training","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04620","citing_title":"Learning to (Learn at Test Time): RNNs with Expressive Hidden States","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12770","citing_title":"WriteSAE: Sparse Autoencoders for Recurrent State","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12770","citing_title":"WriteSAE: Sparse Autoencoders for Recurrent State","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26692","citing_title":"Kimi Linear: An Expressive, Efficient Attention Architecture","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10597","citing_title":"COREY: Entropy-Guided Runtime Chunk Scheduling for Selective Scan Kernels","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06946","citing_title":"Adaptive Memory Decay for Log-Linear Attention","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06169","citing_title":"In-Place Test-Time Training","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05030","citing_title":"Phase-Associative Memory: Sequence Modeling in Complex Hilbert Space","ref_index":97,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EBDWUIG4R5IHRLAHWR2ZNKMJSW","json":"https://pith.science/pith/EBDWUIG4R5IHRLAHWR2ZNKMJSW.json","graph_json":"https://pith.science/api/pith-number/EBDWUIG4R5IHRLAHWR2ZNKMJSW/graph.json","events_json":"https://pith.science/api/pith-number/EBDWUIG4R5IHRLAHWR2ZNKMJSW/events.json","paper":"https://pith.science/paper/EBDWUIG4"},"agent_actions":{"view_html":"https://pith.science/pith/EBDWUIG4R5IHRLAHWR2ZNKMJSW","download_json":"https://pith.science/pith/EBDWUIG4R5IHRLAHWR2ZNKMJSW.json","view_paper":"https://pith.science/paper/EBDWUIG4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.06484&json=true","fetch_graph":"https://pith.science/api/pith-number/EBDWUIG4R5IHRLAHWR2ZNKMJSW/graph.json","fetch_events":"https://pith.science/api/pith-number/EBDWUIG4R5IHRLAHWR2ZNKMJSW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EBDWUIG4R5IHRLAHWR2ZNKMJSW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EBDWUIG4R5IHRLAHWR2ZNKMJSW/action/storage_attestation","attest_author":"https://pith.science/pith/EBDWUIG4R5IHRLAHWR2ZNKMJSW/action/author_attestation","sign_citation":"https://pith.science/pith/EBDWUIG4R5IHRLAHWR2ZNKMJSW/action/citation_signature","submit_replication":"https://pith.science/pith/EBDWUIG4R5IHRLAHWR2ZNKMJSW/action/replication_record"}},"created_at":"2026-07-05T10:01:13.711439+00:00","updated_at":"2026-07-05T10:01:13.711439+00:00"}