{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:D4UEVNNCSWLJSLJ5X24XOW5TUW","short_pith_number":"pith:D4UEVNNC","schema_version":"1.0","canonical_sha256":"1f284ab5a29596992d3dbeb9775bb3a5af0e4c138066c626c2c916cb5a5a1d90","source":{"kind":"arxiv","id":"2503.14456","version":2},"attestation_state":"computed","paper":{"title":"RWKV-7 \"Goose\" with Expressive Dynamic State Evolution","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bo Peng, Christian Zhou-Zheng, Daniel Goldstein, Daniel Wuttke, Eric Alcaide, Guangyu Song, Haowen Hou, Janna Lu, Jiaju Lin, Jiaxing Liu, Johan S. Wind, Kaifeng Tan, Nathan Wilce, Ruichong Zhang, Saiteja Utpala, Tianyi Wu, William Merrill, Xingjian Du","submitted_at":"2025-03-18T17:31:05Z","abstract_excerpt":"We present RWKV-7 \"Goose\", a new sequence modeling architecture with constant memory usage and constant inference time per token. Despite being trained on dramatically fewer tokens than other top models, our 2.9 billion parameter language model achieves a new 3B SoTA on multilingual tasks and matches the current 3B SoTA on English language downstream performance. RWKV-7 introduces a newly generalized formulation of the delta rule with vector-valued gating and in-context learning rates, as well as a relaxed value replacement rule. We show that RWKV-7 can perform state tracking and recognize all"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.14456","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-18T17:31:05Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"940d4d88ec52af7997a2dcc4f33d684cc44adf82e6cdb6604444c03b970f9485","abstract_canon_sha256":"6f7bdae339ac183cd6c014f0e9b45639f1b974f76960224cfb624d77d6f3cdf2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:22.017557Z","signature_b64":"0TdrsVtryC5yX94JEgi+7IHuEunyknnncTq7oWd0ttn1gybNeygD7M0R3HuQlNj5uqtxlWKjw7u96AjZsXhUAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1f284ab5a29596992d3dbeb9775bb3a5af0e4c138066c626c2c916cb5a5a1d90","last_reissued_at":"2026-07-05T10:41:22.017045Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:22.017045Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RWKV-7 \"Goose\" with Expressive Dynamic State Evolution","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bo Peng, Christian Zhou-Zheng, Daniel Goldstein, Daniel Wuttke, Eric Alcaide, Guangyu Song, Haowen Hou, Janna Lu, Jiaju Lin, Jiaxing Liu, Johan S. Wind, Kaifeng Tan, Nathan Wilce, Ruichong Zhang, Saiteja Utpala, Tianyi Wu, William Merrill, Xingjian Du","submitted_at":"2025-03-18T17:31:05Z","abstract_excerpt":"We present RWKV-7 \"Goose\", a new sequence modeling architecture with constant memory usage and constant inference time per token. Despite being trained on dramatically fewer tokens than other top models, our 2.9 billion parameter language model achieves a new 3B SoTA on multilingual tasks and matches the current 3B SoTA on English language downstream performance. RWKV-7 introduces a newly generalized formulation of the delta rule with vector-valued gating and in-context learning rates, as well as a relaxed value replacement rule. We show that RWKV-7 can perform state tracking and recognize all"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.14456","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.14456/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.14456","created_at":"2026-07-05T10:41:22.017102+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.14456v2","created_at":"2026-07-05T10:41:22.017102+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.14456","created_at":"2026-07-05T10:41:22.017102+00:00"},{"alias_kind":"pith_short_12","alias_value":"D4UEVNNCSWLJ","created_at":"2026-07-05T10:41:22.017102+00:00"},{"alias_kind":"pith_short_16","alias_value":"D4UEVNNCSWLJSLJ5","created_at":"2026-07-05T10:41:22.017102+00:00"},{"alias_kind":"pith_short_8","alias_value":"D4UEVNNC","created_at":"2026-07-05T10:41:22.017102+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":31,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.08196","citing_title":"A First-Principles Theory of Slow Thinking and Active Perception","ref_index":126,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05280","citing_title":"Advances in Neural Controlled Differential Equations","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23670","citing_title":"Tapered Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02292","citing_title":"One More Time: Revisiting Neural Quantum States from a Reinforcement Learning Perspective","ref_index":154,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11634","citing_title":"Architecture-Aware Reinforcement Learning Makes Sliding-Window Attention Competitive in Math Reasoning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00902","citing_title":"MG-RWKV: Multi-Grained Context-Aware RWKV for Temporal Forgery Localization","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03825","citing_title":"Dynamic Short Convolutions Improve Transformers","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01765","citing_title":"An Algebraic View of the Expressivity of Recurrent Language Models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28507","citing_title":"Universal Time Series Generation with Neural Controlled Differential Equations","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29453","citing_title":"Forget Less, Generalize More: Unifying Temporal and Structural Adaptation for Dynamic Graphs","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06034","citing_title":"When Good Enough Is Optimal: Multiplication-Only Matrix Inversion Approximation for Quantized Gated DeltaNet","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21016","citing_title":"Gated KalmaNet: A Fading Memory Layer Through Test-Time Ridge Regression","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21070","citing_title":"Towards Understanding Self-Pretraining for Sequence Classification","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06501","citing_title":"Cubit: Token Mixer with Kernel Ridge Regression","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04800","citing_title":"Hybrid Architectures for Language Models: Systematic Analysis and Design Insights","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26083","citing_title":"Nirvana: A Specialized Generalist Model With Task-Aware Memory Mechanism","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2510.27258","citing_title":"Higher-order Linear Attention","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17388","citing_title":"Selective Rotary Position Embedding","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2602.14814","citing_title":"Learning State-Tracking from Code Using Linear RNNs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2602.21204","citing_title":"Test-Time Training with KV Binding Is Secretly Linear Attention","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13473","citing_title":"OSDN: Improving Delta Rule with Provable Online Preconditioning in Linear Attention","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26692","citing_title":"Kimi Linear: An Expressive, Efficient Attention Architecture","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12491","citing_title":"Elastic Attention Cores for Scalable Vision Transformers","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10643","citing_title":"A Single-Layer Model Can Do Language Modeling","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06501","citing_title":"Cubit: Token Mixer with Kernel Ridge Regression","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D4UEVNNCSWLJSLJ5X24XOW5TUW","json":"https://pith.science/pith/D4UEVNNCSWLJSLJ5X24XOW5TUW.json","graph_json":"https://pith.science/api/pith-number/D4UEVNNCSWLJSLJ5X24XOW5TUW/graph.json","events_json":"https://pith.science/api/pith-number/D4UEVNNCSWLJSLJ5X24XOW5TUW/events.json","paper":"https://pith.science/paper/D4UEVNNC"},"agent_actions":{"view_html":"https://pith.science/pith/D4UEVNNCSWLJSLJ5X24XOW5TUW","download_json":"https://pith.science/pith/D4UEVNNCSWLJSLJ5X24XOW5TUW.json","view_paper":"https://pith.science/paper/D4UEVNNC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.14456&json=true","fetch_graph":"https://pith.science/api/pith-number/D4UEVNNCSWLJSLJ5X24XOW5TUW/graph.json","fetch_events":"https://pith.science/api/pith-number/D4UEVNNCSWLJSLJ5X24XOW5TUW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D4UEVNNCSWLJSLJ5X24XOW5TUW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D4UEVNNCSWLJSLJ5X24XOW5TUW/action/storage_attestation","attest_author":"https://pith.science/pith/D4UEVNNCSWLJSLJ5X24XOW5TUW/action/author_attestation","sign_citation":"https://pith.science/pith/D4UEVNNCSWLJSLJ5X24XOW5TUW/action/citation_signature","submit_replication":"https://pith.science/pith/D4UEVNNCSWLJSLJ5X24XOW5TUW/action/replication_record"}},"created_at":"2026-07-05T10:41:22.017102+00:00","updated_at":"2026-07-05T10:41:22.017102+00:00"}