{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6DGONSMB4ALCOVWABZL3UR7KYH","short_pith_number":"pith:6DGONSMB","schema_version":"1.0","canonical_sha256":"f0cce6c981e0162756c00e57ba47eac1ff1228be2270c618b68e42494bb48dca","source":{"kind":"arxiv","id":"2404.07904","version":2},"attestation_state":"computed","paper":{"title":"HGRN2: Gated Linear RNNs with State Expansion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dong Li, Songlin Yang, Weigao Sun, Weixuan Sun, Xuyang Shen, Yiran Zhong, Zhen Qin","submitted_at":"2024-04-11T16:43:03Z","abstract_excerpt":"Hierarchically gated linear RNN (HGRN, \\citealt{HGRN}) has demonstrated competitive training speed and performance in language modeling while offering efficient inference. However, the recurrent state size of HGRN remains relatively small, limiting its expressiveness. To address this issue, we introduce a simple outer product-based state expansion mechanism, which significantly enlarges the recurrent state size without introducing any additional parameters. This enhancement also provides a linear attention interpretation for HGRN2, enabling hardware-efficient training. Our extensive experiment"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.07904","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-11T16:43:03Z","cross_cats_sorted":[],"title_canon_sha256":"59f831808c54844c9b3f90117203e92ab9d19b06a88fb45172ae59b98529badb","abstract_canon_sha256":"d1041a225ab1c02ad922dfa1e4bee3708f13b39ddc4a7cdbce569c24f09f9374"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:56:47.253180Z","signature_b64":"2fLXNZd7ALEOCTyKkq4dqqW0q3qOfvKeQTfcF6qQj/gDGoNjS9BY7atHJNwWu6WbaChQT8y8yu8HRHTsR4JICw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f0cce6c981e0162756c00e57ba47eac1ff1228be2270c618b68e42494bb48dca","last_reissued_at":"2026-07-05T08:56:47.252709Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:56:47.252709Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HGRN2: Gated Linear RNNs with State Expansion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dong Li, Songlin Yang, Weigao Sun, Weixuan Sun, Xuyang Shen, Yiran Zhong, Zhen Qin","submitted_at":"2024-04-11T16:43:03Z","abstract_excerpt":"Hierarchically gated linear RNN (HGRN, \\citealt{HGRN}) has demonstrated competitive training speed and performance in language modeling while offering efficient inference. However, the recurrent state size of HGRN remains relatively small, limiting its expressiveness. To address this issue, we introduce a simple outer product-based state expansion mechanism, which significantly enlarges the recurrent state size without introducing any additional parameters. This enhancement also provides a linear attention interpretation for HGRN2, enabling hardware-efficient training. Our extensive experiment"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.07904","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.07904/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.07904","created_at":"2026-07-05T08:56:47.252767+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.07904v2","created_at":"2026-07-05T08:56:47.252767+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.07904","created_at":"2026-07-05T08:56:47.252767+00:00"},{"alias_kind":"pith_short_12","alias_value":"6DGONSMB4ALC","created_at":"2026-07-05T08:56:47.252767+00:00"},{"alias_kind":"pith_short_16","alias_value":"6DGONSMB4ALCOVWA","created_at":"2026-07-05T08:56:47.252767+00:00"},{"alias_kind":"pith_short_8","alias_value":"6DGONSMB","created_at":"2026-07-05T08:56:47.252767+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.16533","citing_title":"Kairos: A Regret-Aware Native World-Action Model Stack for Physical AI","ref_index":185,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11634","citing_title":"Architecture-Aware Reinforcement Learning Makes Sliding-Window Attention Competitive in Math Reasoning","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03825","citing_title":"Dynamic Short Convolutions Improve Transformers","ref_index":174,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31163","citing_title":"Memory by Design: Probabilistic Sequence Layers","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20670","citing_title":"LT2: Linear-Time Looped Transformers","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30562","citing_title":"Morphing into Hybrid Attention Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28507","citing_title":"Universal Time Series Generation with Neural Controlled Differential Equations","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22791","citing_title":"Gated DeltaNet-2: Decoupling Erase and Write in Linear Attention","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2603.15031","citing_title":"Attention Residuals","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20670","citing_title":"LT2: Linear-Time Looped Transformers","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06501","citing_title":"Cubit: Token Mixer with Kernel Ridge Regression","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26083","citing_title":"Nirvana: A Specialized Generalist Model With Task-Aware Memory Mechanism","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17388","citing_title":"Selective Rotary Position Embedding","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2602.21204","citing_title":"Test-Time Training with KV Binding Is Secretly Linear Attention","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2312.06635","citing_title":"Gated Linear Attention Transformers with Hardware-Efficient Training","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12992","citing_title":"SpikeProphecy: A Large-Scale Benchmark for Autoregressive Neural Population Forecasting","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26692","citing_title":"Kimi Linear: An Expressive, Efficient Attention Architecture","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12491","citing_title":"Elastic Attention Cores for Scalable Vision Transformers","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06501","citing_title":"Cubit: Token Mixer with Kernel Ridge Regression","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05066","citing_title":"The Impossibility Triangle of Long-Context Modeling","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19021","citing_title":"FG$^2$-GDN: Enhancing Long-Context Gated Delta Networks with Doubly Fine-Grained Control","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2405.21060","citing_title":"Transformers are SSMs: Generalized Models and Efficient Algorithms Through Structured State Space Duality","ref_index":84,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6DGONSMB4ALCOVWABZL3UR7KYH","json":"https://pith.science/pith/6DGONSMB4ALCOVWABZL3UR7KYH.json","graph_json":"https://pith.science/api/pith-number/6DGONSMB4ALCOVWABZL3UR7KYH/graph.json","events_json":"https://pith.science/api/pith-number/6DGONSMB4ALCOVWABZL3UR7KYH/events.json","paper":"https://pith.science/paper/6DGONSMB"},"agent_actions":{"view_html":"https://pith.science/pith/6DGONSMB4ALCOVWABZL3UR7KYH","download_json":"https://pith.science/pith/6DGONSMB4ALCOVWABZL3UR7KYH.json","view_paper":"https://pith.science/paper/6DGONSMB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.07904&json=true","fetch_graph":"https://pith.science/api/pith-number/6DGONSMB4ALCOVWABZL3UR7KYH/graph.json","fetch_events":"https://pith.science/api/pith-number/6DGONSMB4ALCOVWABZL3UR7KYH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6DGONSMB4ALCOVWABZL3UR7KYH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6DGONSMB4ALCOVWABZL3UR7KYH/action/storage_attestation","attest_author":"https://pith.science/pith/6DGONSMB4ALCOVWABZL3UR7KYH/action/author_attestation","sign_citation":"https://pith.science/pith/6DGONSMB4ALCOVWABZL3UR7KYH/action/citation_signature","submit_replication":"https://pith.science/pith/6DGONSMB4ALCOVWABZL3UR7KYH/action/replication_record"}},"created_at":"2026-07-05T08:56:47.252767+00:00","updated_at":"2026-07-05T08:56:47.252767+00:00"}