{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:3U6YOJJIZHACAFZNJAZXGFQYRY","short_pith_number":"pith:3U6YOJJI","schema_version":"1.0","canonical_sha256":"dd3d872528c9c020172d48337316188e0d9c7ecc69549c1b70d9d8e22604ebef","source":{"kind":"arxiv","id":"2103.13076","version":2},"attestation_state":"computed","paper":{"title":"Finetuning Pretrained Transformers into RNNs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dani Yogatama, Gabriel Ilharco, Hao Peng, Jungo Kasai, Nikolaos Pappas, Noah A. Smith, Weizhu Chen, Yi Mao, Yizhe Zhang","submitted_at":"2021-03-24T10:50:43Z","abstract_excerpt":"Transformers have outperformed recurrent neural networks (RNNs) in natural language generation. But this comes with a significant computational cost, as the attention mechanism's complexity scales quadratically with sequence length. Efficient transformer variants have received increasing interest in recent works. Among them, a linear-complexity recurrent variant has proven well suited for autoregressive generation. It approximates the softmax attention with randomized or heuristic feature maps, but can be difficult to train and may yield suboptimal accuracy. This work aims to convert a pretrai"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2103.13076","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-03-24T10:50:43Z","cross_cats_sorted":[],"title_canon_sha256":"e1f4136ca6c9b2667250fa1685b69eef261961f7ac933150fda4074379986a5c","abstract_canon_sha256":"a122b35384b1987d9e6b77c4e3478cf18fa3a1630289ae04f0a3af11d04bb0df"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:15:32.187852Z","signature_b64":"a0cHryxmlYxntpUkQQ8b0mAlBqkSPJ1OrJXgGEqRIwyauJBoARR97+6guRveAKtRm+CICUI//csqv3le8y7VBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dd3d872528c9c020172d48337316188e0d9c7ecc69549c1b70d9d8e22604ebef","last_reissued_at":"2026-07-05T03:15:32.187396Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:15:32.187396Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Finetuning Pretrained Transformers into RNNs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dani Yogatama, Gabriel Ilharco, Hao Peng, Jungo Kasai, Nikolaos Pappas, Noah A. Smith, Weizhu Chen, Yi Mao, Yizhe Zhang","submitted_at":"2021-03-24T10:50:43Z","abstract_excerpt":"Transformers have outperformed recurrent neural networks (RNNs) in natural language generation. But this comes with a significant computational cost, as the attention mechanism's complexity scales quadratically with sequence length. Efficient transformer variants have received increasing interest in recent works. Among them, a linear-complexity recurrent variant has proven well suited for autoregressive generation. It approximates the softmax attention with randomized or heuristic feature maps, but can be difficult to train and may yield suboptimal accuracy. This work aims to convert a pretrai"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.13076","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.13076/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2103.13076","created_at":"2026-07-05T03:15:32.187461+00:00"},{"alias_kind":"arxiv_version","alias_value":"2103.13076v2","created_at":"2026-07-05T03:15:32.187461+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.13076","created_at":"2026-07-05T03:15:32.187461+00:00"},{"alias_kind":"pith_short_12","alias_value":"3U6YOJJIZHAC","created_at":"2026-07-05T03:15:32.187461+00:00"},{"alias_kind":"pith_short_16","alias_value":"3U6YOJJIZHACAFZN","created_at":"2026-07-05T03:15:32.187461+00:00"},{"alias_kind":"pith_short_8","alias_value":"3U6YOJJI","created_at":"2026-07-05T03:15:32.187461+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06479","citing_title":"Pretraining Recurrent Networks without Recurrence","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2410.13846","citing_title":"LightTransfer: Your Long-Context LLM is Secretly a Hybrid Model with Effortless Adaptation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2402.17762","citing_title":"Massive Activations in Large Language Models","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14191","citing_title":"Attention to Mamba: A Recipe for Cross-Architecture Distillation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2401.10774","citing_title":"Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads","ref_index":242,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3U6YOJJIZHACAFZNJAZXGFQYRY","json":"https://pith.science/pith/3U6YOJJIZHACAFZNJAZXGFQYRY.json","graph_json":"https://pith.science/api/pith-number/3U6YOJJIZHACAFZNJAZXGFQYRY/graph.json","events_json":"https://pith.science/api/pith-number/3U6YOJJIZHACAFZNJAZXGFQYRY/events.json","paper":"https://pith.science/paper/3U6YOJJI"},"agent_actions":{"view_html":"https://pith.science/pith/3U6YOJJIZHACAFZNJAZXGFQYRY","download_json":"https://pith.science/pith/3U6YOJJIZHACAFZNJAZXGFQYRY.json","view_paper":"https://pith.science/paper/3U6YOJJI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2103.13076&json=true","fetch_graph":"https://pith.science/api/pith-number/3U6YOJJIZHACAFZNJAZXGFQYRY/graph.json","fetch_events":"https://pith.science/api/pith-number/3U6YOJJIZHACAFZNJAZXGFQYRY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3U6YOJJIZHACAFZNJAZXGFQYRY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3U6YOJJIZHACAFZNJAZXGFQYRY/action/storage_attestation","attest_author":"https://pith.science/pith/3U6YOJJIZHACAFZNJAZXGFQYRY/action/author_attestation","sign_citation":"https://pith.science/pith/3U6YOJJIZHACAFZNJAZXGFQYRY/action/citation_signature","submit_replication":"https://pith.science/pith/3U6YOJJIZHACAFZNJAZXGFQYRY/action/replication_record"}},"created_at":"2026-07-05T03:15:32.187461+00:00","updated_at":"2026-07-05T03:15:32.187461+00:00"}