{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:V6ZW3GQCCMUNOOC36NK2HVUZHV","short_pith_number":"pith:V6ZW3GQC","schema_version":"1.0","canonical_sha256":"afb36d9a021328d7385bf355a3d6993d4cbf0cdb6ae42d241a37deab5d730669","source":{"kind":"arxiv","id":"2501.15570","version":1},"attestation_state":"computed","paper":{"title":"ARWKV: Pretrain is not what we need, an RNN-Attention-Based Language Model Born from Transformer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Lin Yueyu, Liu Xiao, Li Zhiyuan, Peter Yue","submitted_at":"2025-01-26T15:56:56Z","abstract_excerpt":"As is known, hybrid quadratic and subquadratic attention models in multi-head architectures have surpassed both Transformer and Linear RNN models , with these works primarily focusing on reducing KV complexity and improving efficiency. For further research on expressiveness, we introduce our series of models distilled from Qwen 2.5, based on pure native RWKV-7 attention, which aims to make RNN more expressive and demonstrates state tracking ability beyond transformers. We work with QRWK 32B based on RWKV-6 architecture, another approach that reduces the entire knowledge processing time to just"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.15570","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-01-26T15:56:56Z","cross_cats_sorted":[],"title_canon_sha256":"c18022e2581b510b79371e7854c894b1a56fad3ec80d479a7cc5b30654acbdac","abstract_canon_sha256":"2b4c68c29dddc9568f4d8c61df1c6114f67eba9a68123282d5a925e702ecaf25"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:05:31.230437Z","signature_b64":"nVjq1dRExTISrbfsAnINZhkC/50OWCoA6QFGlXwQdwHlPeVx3YKkni5twvOCHUkIpHeO8BqX5FCvcZklWEtICA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"afb36d9a021328d7385bf355a3d6993d4cbf0cdb6ae42d241a37deab5d730669","last_reissued_at":"2026-07-05T10:05:31.229929Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:05:31.229929Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ARWKV: Pretrain is not what we need, an RNN-Attention-Based Language Model Born from Transformer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Lin Yueyu, Liu Xiao, Li Zhiyuan, Peter Yue","submitted_at":"2025-01-26T15:56:56Z","abstract_excerpt":"As is known, hybrid quadratic and subquadratic attention models in multi-head architectures have surpassed both Transformer and Linear RNN models , with these works primarily focusing on reducing KV complexity and improving efficiency. For further research on expressiveness, we introduce our series of models distilled from Qwen 2.5, based on pure native RWKV-7 attention, which aims to make RNN more expressive and demonstrates state tracking ability beyond transformers. We work with QRWK 32B based on RWKV-6 architecture, another approach that reduces the entire knowledge processing time to just"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.15570","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.15570/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.15570","created_at":"2026-07-05T10:05:31.229988+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.15570v1","created_at":"2026-07-05T10:05:31.229988+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.15570","created_at":"2026-07-05T10:05:31.229988+00:00"},{"alias_kind":"pith_short_12","alias_value":"V6ZW3GQCCMUN","created_at":"2026-07-05T10:05:31.229988+00:00"},{"alias_kind":"pith_short_16","alias_value":"V6ZW3GQCCMUNOOC3","created_at":"2026-07-05T10:05:31.229988+00:00"},{"alias_kind":"pith_short_8","alias_value":"V6ZW3GQC","created_at":"2026-07-05T10:05:31.229988+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.19191","citing_title":"WuNeng: Hybrid State with Attention","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V6ZW3GQCCMUNOOC36NK2HVUZHV","json":"https://pith.science/pith/V6ZW3GQCCMUNOOC36NK2HVUZHV.json","graph_json":"https://pith.science/api/pith-number/V6ZW3GQCCMUNOOC36NK2HVUZHV/graph.json","events_json":"https://pith.science/api/pith-number/V6ZW3GQCCMUNOOC36NK2HVUZHV/events.json","paper":"https://pith.science/paper/V6ZW3GQC"},"agent_actions":{"view_html":"https://pith.science/pith/V6ZW3GQCCMUNOOC36NK2HVUZHV","download_json":"https://pith.science/pith/V6ZW3GQCCMUNOOC36NK2HVUZHV.json","view_paper":"https://pith.science/paper/V6ZW3GQC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.15570&json=true","fetch_graph":"https://pith.science/api/pith-number/V6ZW3GQCCMUNOOC36NK2HVUZHV/graph.json","fetch_events":"https://pith.science/api/pith-number/V6ZW3GQCCMUNOOC36NK2HVUZHV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V6ZW3GQCCMUNOOC36NK2HVUZHV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V6ZW3GQCCMUNOOC36NK2HVUZHV/action/storage_attestation","attest_author":"https://pith.science/pith/V6ZW3GQCCMUNOOC36NK2HVUZHV/action/author_attestation","sign_citation":"https://pith.science/pith/V6ZW3GQCCMUNOOC36NK2HVUZHV/action/citation_signature","submit_replication":"https://pith.science/pith/V6ZW3GQCCMUNOOC36NK2HVUZHV/action/replication_record"}},"created_at":"2026-07-05T10:05:31.229988+00:00","updated_at":"2026-07-05T10:05:31.229988+00:00"}