{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7VCS4PN7CG5SU5PTHSAT5RVCU2","short_pith_number":"pith:7VCS4PN7","schema_version":"1.0","canonical_sha256":"fd452e3dbf11bb2a75f33c813ec6a2a6bb7e9b877f68f290251f94325443d263","source":{"kind":"arxiv","id":"2402.18510","version":4},"attestation_state":"computed","paper":{"title":"RNNs are not Transformers (Yet): The Key Bottleneck on In-context Retrieval","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Kaifeng Lyu, Kaiyue Wen, Xingyu Dang","submitted_at":"2024-02-28T17:38:06Z","abstract_excerpt":"This paper investigates the gap in representation powers of Recurrent Neural Networks (RNNs) and Transformers in the context of solving algorithmic problems. We focus on understanding whether RNNs, known for their memory efficiency in handling long sequences, can match the performance of Transformers, particularly when enhanced with Chain-of-Thought (CoT) prompting. Our theoretical analysis reveals that CoT improves RNNs but is insufficient to close the gap with Transformers. A key bottleneck lies in the inability of RNNs to perfectly retrieve information from the context, even with CoT: for s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.18510","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-28T17:38:06Z","cross_cats_sorted":["cs.CL","stat.ML"],"title_canon_sha256":"eaf30337ec3de47b78d57493674a3872973547d299a02c5d4df220cd177de4c6","abstract_canon_sha256":"fb7fc42601ddc63639f2d32589f3be604132ff0c69a0af3f2f6d25c850154e62"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:44.969617Z","signature_b64":"1K07ltVIPzH6u4HU4RR5/jrad+aLtfMchcB1coaTgIICamrF37tE/o4sR3AVVdz7cV9XjG7cele/nEuTmR07DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fd452e3dbf11bb2a75f33c813ec6a2a6bb7e9b877f68f290251f94325443d263","last_reissued_at":"2026-07-05T09:45:44.969167Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:44.969167Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RNNs are not Transformers (Yet): The Key Bottleneck on In-context Retrieval","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Kaifeng Lyu, Kaiyue Wen, Xingyu Dang","submitted_at":"2024-02-28T17:38:06Z","abstract_excerpt":"This paper investigates the gap in representation powers of Recurrent Neural Networks (RNNs) and Transformers in the context of solving algorithmic problems. We focus on understanding whether RNNs, known for their memory efficiency in handling long sequences, can match the performance of Transformers, particularly when enhanced with Chain-of-Thought (CoT) prompting. Our theoretical analysis reveals that CoT improves RNNs but is insufficient to close the gap with Transformers. A key bottleneck lies in the inability of RNNs to perfectly retrieve information from the context, even with CoT: for s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.18510","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.18510/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.18510","created_at":"2026-07-05T09:45:44.969219+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.18510v4","created_at":"2026-07-05T09:45:44.969219+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.18510","created_at":"2026-07-05T09:45:44.969219+00:00"},{"alias_kind":"pith_short_12","alias_value":"7VCS4PN7CG5S","created_at":"2026-07-05T09:45:44.969219+00:00"},{"alias_kind":"pith_short_16","alias_value":"7VCS4PN7CG5SU5PT","created_at":"2026-07-05T09:45:44.969219+00:00"},{"alias_kind":"pith_short_8","alias_value":"7VCS4PN7","created_at":"2026-07-05T09:45:44.969219+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":123,"is_internal_anchor":true},{"citing_arxiv_id":"2605.13473","citing_title":"OSDN: Improving Delta Rule with Provable Online Preconditioning in Linear Attention","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26692","citing_title":"Kimi Linear: An Expressive, Efficient Attention Architecture","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05838","citing_title":"MDN: Parallelizing Stepwise Momentum for Delta Linear Attention","ref_index":83,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7VCS4PN7CG5SU5PTHSAT5RVCU2","json":"https://pith.science/pith/7VCS4PN7CG5SU5PTHSAT5RVCU2.json","graph_json":"https://pith.science/api/pith-number/7VCS4PN7CG5SU5PTHSAT5RVCU2/graph.json","events_json":"https://pith.science/api/pith-number/7VCS4PN7CG5SU5PTHSAT5RVCU2/events.json","paper":"https://pith.science/paper/7VCS4PN7"},"agent_actions":{"view_html":"https://pith.science/pith/7VCS4PN7CG5SU5PTHSAT5RVCU2","download_json":"https://pith.science/pith/7VCS4PN7CG5SU5PTHSAT5RVCU2.json","view_paper":"https://pith.science/paper/7VCS4PN7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.18510&json=true","fetch_graph":"https://pith.science/api/pith-number/7VCS4PN7CG5SU5PTHSAT5RVCU2/graph.json","fetch_events":"https://pith.science/api/pith-number/7VCS4PN7CG5SU5PTHSAT5RVCU2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7VCS4PN7CG5SU5PTHSAT5RVCU2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7VCS4PN7CG5SU5PTHSAT5RVCU2/action/storage_attestation","attest_author":"https://pith.science/pith/7VCS4PN7CG5SU5PTHSAT5RVCU2/action/author_attestation","sign_citation":"https://pith.science/pith/7VCS4PN7CG5SU5PTHSAT5RVCU2/action/citation_signature","submit_replication":"https://pith.science/pith/7VCS4PN7CG5SU5PTHSAT5RVCU2/action/replication_record"}},"created_at":"2026-07-05T09:45:44.969219+00:00","updated_at":"2026-07-05T09:45:44.969219+00:00"}