{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LHIA7KCLDIDBHGVSIAXAPLPAA4","short_pith_number":"pith:LHIA7KCL","schema_version":"1.0","canonical_sha256":"59d00fa84b1a06139ab2402e07ade007057fee9de1242ebc52b55cea56bb01c7","source":{"kind":"arxiv","id":"2410.01201","version":3},"attestation_state":"computed","paper":{"title":"Were RNNs All We Needed?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Frederick Tung, Hossein Hajimirsadeghi, Leo Feng, Mohamed Osama Ahmed, Yoshua Bengio","submitted_at":"2024-10-02T03:06:49Z","abstract_excerpt":"The introduction of Transformers in 2017 reshaped the landscape of deep learning. Originally proposed for sequence modelling, Transformers have since achieved widespread success across various domains. However, the scalability limitations of Transformers - particularly with respect to sequence length - have sparked renewed interest in novel recurrent models that are parallelizable during training, offer comparable performance, and scale more effectively. In this work, we revisit sequence modelling from a historical perspective, focusing on Recurrent Neural Networks (RNNs), which dominated the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.01201","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-02T03:06:49Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f8aa06d258bb04dca2746f89fe193da0012bcec135ddd611e67283a6eee01dc6","abstract_canon_sha256":"5b05aaf3e8e3dec9178e39f9992c760780735253acf105266e37f827e65d1c8a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:42.624671Z","signature_b64":"7QwXGHVYwKoUIY5AKDyhCLQqTuLZb/WyJikvVTdOzWXrzUlXvP9EY3YmQamYXt6AjWbkzB/ZVLZLAtd7jDRiAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"59d00fa84b1a06139ab2402e07ade007057fee9de1242ebc52b55cea56bb01c7","last_reissued_at":"2026-07-05T09:41:42.624177Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:42.624177Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Were RNNs All We Needed?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Frederick Tung, Hossein Hajimirsadeghi, Leo Feng, Mohamed Osama Ahmed, Yoshua Bengio","submitted_at":"2024-10-02T03:06:49Z","abstract_excerpt":"The introduction of Transformers in 2017 reshaped the landscape of deep learning. Originally proposed for sequence modelling, Transformers have since achieved widespread success across various domains. However, the scalability limitations of Transformers - particularly with respect to sequence length - have sparked renewed interest in novel recurrent models that are parallelizable during training, offer comparable performance, and scale more effectively. In this work, we revisit sequence modelling from a historical perspective, focusing on Recurrent Neural Networks (RNNs), which dominated the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.01201","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.01201/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.01201","created_at":"2026-07-05T09:41:42.624237+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.01201v3","created_at":"2026-07-05T09:41:42.624237+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.01201","created_at":"2026-07-05T09:41:42.624237+00:00"},{"alias_kind":"pith_short_12","alias_value":"LHIA7KCLDIDB","created_at":"2026-07-05T09:41:42.624237+00:00"},{"alias_kind":"pith_short_16","alias_value":"LHIA7KCLDIDBHGVS","created_at":"2026-07-05T09:41:42.624237+00:00"},{"alias_kind":"pith_short_8","alias_value":"LHIA7KCL","created_at":"2026-07-05T09:41:42.624237+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23332","citing_title":"Don't Listen to Me: A Lightweight, Low-Latency Model for Own-Voice Cancellation in Far-Field Speech Enhancement","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06479","citing_title":"Pretraining Recurrent Networks without Recurrence","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31110","citing_title":"Building Generalization Into Behavior Generation Via Adaptive Compositions of Regularities","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31367","citing_title":"Trading Complexity for Expressivity Through Structured Generalized Linear Token Mixing","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15216","citing_title":"Hardware-Software Co-Design of Scalable, Energy-Efficient Analog Recurrent Computations","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30777","citing_title":"Unveiling Transferability in Trajectory Prediction via Latent Scene Embeddings","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15216","citing_title":"Hardware-Software Co-Design of Scalable, Energy-Efficient Analog Recurrent Computations","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13807","citing_title":"Parallel Scan Recurrent Neural Quantum States for Scalable Variational Monte Carlo","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11855","citing_title":"Improving the Performance and Learning Stability of Parallelizable RNNs Designed for Ultra-Low Power Applications","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12206","citing_title":"On the Importance of Multistability for Horizon Generalization in Reinforcement Learning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05066","citing_title":"The Impossibility Triangle of Long-Context Modeling","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19343","citing_title":"Scalable Memristive-Friendly Reservoir Computing for Time Series Classification","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LHIA7KCLDIDBHGVSIAXAPLPAA4","json":"https://pith.science/pith/LHIA7KCLDIDBHGVSIAXAPLPAA4.json","graph_json":"https://pith.science/api/pith-number/LHIA7KCLDIDBHGVSIAXAPLPAA4/graph.json","events_json":"https://pith.science/api/pith-number/LHIA7KCLDIDBHGVSIAXAPLPAA4/events.json","paper":"https://pith.science/paper/LHIA7KCL"},"agent_actions":{"view_html":"https://pith.science/pith/LHIA7KCLDIDBHGVSIAXAPLPAA4","download_json":"https://pith.science/pith/LHIA7KCLDIDBHGVSIAXAPLPAA4.json","view_paper":"https://pith.science/paper/LHIA7KCL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.01201&json=true","fetch_graph":"https://pith.science/api/pith-number/LHIA7KCLDIDBHGVSIAXAPLPAA4/graph.json","fetch_events":"https://pith.science/api/pith-number/LHIA7KCLDIDBHGVSIAXAPLPAA4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LHIA7KCLDIDBHGVSIAXAPLPAA4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LHIA7KCLDIDBHGVSIAXAPLPAA4/action/storage_attestation","attest_author":"https://pith.science/pith/LHIA7KCLDIDBHGVSIAXAPLPAA4/action/author_attestation","sign_citation":"https://pith.science/pith/LHIA7KCLDIDBHGVSIAXAPLPAA4/action/citation_signature","submit_replication":"https://pith.science/pith/LHIA7KCLDIDBHGVSIAXAPLPAA4/action/replication_record"}},"created_at":"2026-07-05T09:41:42.624237+00:00","updated_at":"2026-07-05T09:41:42.624237+00:00"}