{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:H4UI5KPABQ67UHIU7PFOETSZTM","short_pith_number":"pith:H4UI5KPA","schema_version":"1.0","canonical_sha256":"3f288ea9e00c3dfa1d14fbcae24e599b06345efd02b542e36187bc20b73d3ae9","source":{"kind":"arxiv","id":"2303.06349","version":1},"attestation_state":"computed","paper":{"title":"Resurrecting Recurrent Neural Networks for Long Sequences","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Albert Gu, Antonio Orvieto, Anushan Fernando, Caglar Gulcehre, Razvan Pascanu, Samuel L Smith, Soham De","submitted_at":"2023-03-11T08:53:11Z","abstract_excerpt":"Recurrent Neural Networks (RNNs) offer fast inference on long sequences but are hard to optimize and slow to train. Deep state-space models (SSMs) have recently been shown to perform remarkably well on long sequence modeling tasks, and have the added benefits of fast parallelizable training and RNN-like fast inference. However, while SSMs are superficially similar to RNNs, there are important differences that make it unclear where their performance boost over RNNs comes from. In this paper, we show that careful design of deep RNNs using standard signal propagation arguments can recover the imp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.06349","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-03-11T08:53:11Z","cross_cats_sorted":[],"title_canon_sha256":"d72983c372f189bce6f3403b19150630748ccd909dc5ed37d2580aa4ffb1000b","abstract_canon_sha256":"b7c931a671f2eb6aa0f7ab88a39c0d8b903e0ab102bc971f71066d84a22ea76a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:50:17.732868Z","signature_b64":"jK2qt8Xz1JtZ/AgR5Zu+Bky3/7rf9W8IvdqPyfvZ9HAb0pVEk9hiCagzDn81rhSKUTDVtq0QG+dB/v4mdsBKAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3f288ea9e00c3dfa1d14fbcae24e599b06345efd02b542e36187bc20b73d3ae9","last_reissued_at":"2026-07-05T05:50:17.732502Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:50:17.732502Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Resurrecting Recurrent Neural Networks for Long Sequences","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Albert Gu, Antonio Orvieto, Anushan Fernando, Caglar Gulcehre, Razvan Pascanu, Samuel L Smith, Soham De","submitted_at":"2023-03-11T08:53:11Z","abstract_excerpt":"Recurrent Neural Networks (RNNs) offer fast inference on long sequences but are hard to optimize and slow to train. Deep state-space models (SSMs) have recently been shown to perform remarkably well on long sequence modeling tasks, and have the added benefits of fast parallelizable training and RNN-like fast inference. However, while SSMs are superficially similar to RNNs, there are important differences that make it unclear where their performance boost over RNNs comes from. In this paper, we show that careful design of deep RNNs using standard signal propagation arguments can recover the imp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.06349","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.06349/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.06349","created_at":"2026-07-05T05:50:17.732558+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.06349v1","created_at":"2026-07-05T05:50:17.732558+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.06349","created_at":"2026-07-05T05:50:17.732558+00:00"},{"alias_kind":"pith_short_12","alias_value":"H4UI5KPABQ67","created_at":"2026-07-05T05:50:17.732558+00:00"},{"alias_kind":"pith_short_16","alias_value":"H4UI5KPABQ67UHIU","created_at":"2026-07-05T05:50:17.732558+00:00"},{"alias_kind":"pith_short_8","alias_value":"H4UI5KPA","created_at":"2026-07-05T05:50:17.732558+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25010","citing_title":"Emergent Capabilities Arise Randomly from Learning Sparse Attention Patterns","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23332","citing_title":"Don't Listen to Me: A Lightweight, Low-Latency Model for Own-Voice Cancellation in Far-Field Speech Enhancement","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15216","citing_title":"Hardware-Software Co-Design of Scalable, Energy-Efficient Analog Recurrent Computations","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24709","citing_title":"Streaming Reinforcement Learning under Partial Observability with Real-Time Recurrent Learning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11049","citing_title":"Free Parametrization of L_2-Bounded Structured State-Space Controllers for Nonlinear Control with Stability Guarantees","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2503.23818","citing_title":"L2RU: a Structured State Space Model with prescribed L2-bound","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15216","citing_title":"Hardware-Software Co-Design of Scalable, Energy-Efficient Analog Recurrent Computations","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2506.06374","citing_title":"SiLIF: Structured State Space Model Dynamics and Parametrization for Spiking Neural Networks","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2507.01829","citing_title":"mGRADE: Minimal Recurrent Gating Meets Delay Convolutions for Lightweight Sequence Modeling","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13807","citing_title":"Parallel Scan Recurrent Neural Quantum States for Scalable Variational Monte Carlo","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2308.14508","citing_title":"LongBench: A Bilingual, Multitask Benchmark for Long Context Understanding","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2307.08621","citing_title":"Retentive Network: A Successor to Transformer for Large Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19343","citing_title":"Scalable Memristive-Friendly Reservoir Computing for Time Series Classification","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H4UI5KPABQ67UHIU7PFOETSZTM","json":"https://pith.science/pith/H4UI5KPABQ67UHIU7PFOETSZTM.json","graph_json":"https://pith.science/api/pith-number/H4UI5KPABQ67UHIU7PFOETSZTM/graph.json","events_json":"https://pith.science/api/pith-number/H4UI5KPABQ67UHIU7PFOETSZTM/events.json","paper":"https://pith.science/paper/H4UI5KPA"},"agent_actions":{"view_html":"https://pith.science/pith/H4UI5KPABQ67UHIU7PFOETSZTM","download_json":"https://pith.science/pith/H4UI5KPABQ67UHIU7PFOETSZTM.json","view_paper":"https://pith.science/paper/H4UI5KPA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.06349&json=true","fetch_graph":"https://pith.science/api/pith-number/H4UI5KPABQ67UHIU7PFOETSZTM/graph.json","fetch_events":"https://pith.science/api/pith-number/H4UI5KPABQ67UHIU7PFOETSZTM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H4UI5KPABQ67UHIU7PFOETSZTM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H4UI5KPABQ67UHIU7PFOETSZTM/action/storage_attestation","attest_author":"https://pith.science/pith/H4UI5KPABQ67UHIU7PFOETSZTM/action/author_attestation","sign_citation":"https://pith.science/pith/H4UI5KPABQ67UHIU7PFOETSZTM/action/citation_signature","submit_replication":"https://pith.science/pith/H4UI5KPABQ67UHIU7PFOETSZTM/action/replication_record"}},"created_at":"2026-07-05T05:50:17.732558+00:00","updated_at":"2026-07-05T05:50:17.732558+00:00"}