{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:GM5CED6SAUEGHCZQTLULNMOGTS","short_pith_number":"pith:GM5CED6S","schema_version":"1.0","canonical_sha256":"333a220fd20508638b309ae8b6b1c69c98299e54aea2c276fc53433320e19393","source":{"kind":"arxiv","id":"2212.14052","version":3},"attestation_state":"computed","paper":{"title":"Hungry Hungry Hippos: Towards Language Modeling with State Space Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Armin W. Thomas, Atri Rudra, Christopher R\\'e, Daniel Y. Fu, Khaled K. Saab, Tri Dao","submitted_at":"2022-12-28T17:56:03Z","abstract_excerpt":"State space models (SSMs) have demonstrated state-of-the-art sequence modeling performance in some modalities, but underperform attention in language modeling. Moreover, despite scaling nearly linearly in sequence length instead of quadratically, SSMs are still slower than Transformers due to poor hardware utilization. In this paper, we make progress on understanding the expressivity gap between SSMs and attention in language modeling, and on reducing the hardware barrier between SSMs and attention. First, we use synthetic language modeling tasks to understand the gap between SSMs and attentio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.14052","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-12-28T17:56:03Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"326a02f7fcf837eded683d30bb4c850ab0843e129dab6f98ebe04a7fbe316f17","abstract_canon_sha256":"2714d54d104fe5a7143a7f30f660c71b210e24feec200a7cb0c40e699e423800"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:05:32.689075Z","signature_b64":"bFLjLOuJ1MT954ieO9zmVnqFFotVawmAb23JdHfmSBxuIaksbPJeYnATYolb04y9Xv4kC7o8PpLmqtijWPDyAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"333a220fd20508638b309ae8b6b1c69c98299e54aea2c276fc53433320e19393","last_reissued_at":"2026-07-05T06:05:32.688631Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:05:32.688631Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hungry Hungry Hippos: Towards Language Modeling with State Space Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Armin W. Thomas, Atri Rudra, Christopher R\\'e, Daniel Y. Fu, Khaled K. Saab, Tri Dao","submitted_at":"2022-12-28T17:56:03Z","abstract_excerpt":"State space models (SSMs) have demonstrated state-of-the-art sequence modeling performance in some modalities, but underperform attention in language modeling. Moreover, despite scaling nearly linearly in sequence length instead of quadratically, SSMs are still slower than Transformers due to poor hardware utilization. In this paper, we make progress on understanding the expressivity gap between SSMs and attention in language modeling, and on reducing the hardware barrier between SSMs and attention. First, we use synthetic language modeling tasks to understand the gap between SSMs and attentio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.14052","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.14052/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.14052","created_at":"2026-07-05T06:05:32.688678+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.14052v3","created_at":"2026-07-05T06:05:32.688678+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.14052","created_at":"2026-07-05T06:05:32.688678+00:00"},{"alias_kind":"pith_short_12","alias_value":"GM5CED6SAUEG","created_at":"2026-07-05T06:05:32.688678+00:00"},{"alias_kind":"pith_short_16","alias_value":"GM5CED6SAUEGHCZQ","created_at":"2026-07-05T06:05:32.688678+00:00"},{"alias_kind":"pith_short_8","alias_value":"GM5CED6S","created_at":"2026-07-05T06:05:32.688678+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":33,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07386","citing_title":"Sparse Delta Memory: Scaling the State of Linear RNNs through Sparsity","ref_index":90,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24320","citing_title":"ZONOS2 Technical Report","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24969","citing_title":"Frequency Domain Reservoir Computing","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26749","citing_title":"Structure Before Collapse: Transient semantic geometry in next-token prediction","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19932","citing_title":"Spatial-Aware Reduction Framework: Towards Efficient and Faithful Visual State Space Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12895","citing_title":"LongSpike: Fractional Order Spiking State Space Models for Efficient Long Sequence Learning","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02385","citing_title":"How Optimality Structures Sparse Dictionaries: A Theory for Understanding SAE Representations","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09862","citing_title":"Blurry Window Attention","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24320","citing_title":"ZONOS2 Technical Report","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08696","citing_title":"Structured Recurrent Mixers for Massively Parallelized Sequence Generation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30030","citing_title":"CogSENet: Blind Image Deblurring with Blur-Conditioned Semantic Routing and Explicit Frequency Fusion","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26797","citing_title":"Latent Recurrent Transformer: Architecture Exploration, Training Strategies, and Scaling Behavior","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2411.18328","citing_title":"EventCrab: Harnessing Frame and Point Synergy for Event-based Action Recognition and Beyond","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2503.18970","citing_title":"Advancing Intelligent Sequence Modeling: Evolution, Trade-offs, and Applications of State- Space Architectures from S4 to Mamba","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08696","citing_title":"Structured Recurrent Mixers for Massively Parallelized Sequence Generation","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09110","citing_title":"CodeBrain: Bridging Decoupled Tokenizer and Multi-Scale Architecture for EEG Foundation Model","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2507.01829","citing_title":"mGRADE: Minimal Recurrent Gating Meets Delay Convolutions for Lightweight Sequence Modeling","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24552","citing_title":"Short window attention enables long-term memorization","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04800","citing_title":"Hybrid Architectures for Language Models: Systematic Analysis and Design Insights","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12602","citing_title":"Exact Flow Linear Attention: Exact Solution from Continuous-Time Dynamics","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2602.20730","citing_title":"Rethinking Efficiency in Neural Combinatorial Optimization: Batched Preference Optimization with Mamba","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14360","citing_title":"M$^2$RNN: Non-Linear RNNs with Matrix-Valued States for Scalable Language Modeling","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2402.19427","citing_title":"Griffin: Mixing Gated Linear Recurrences with Local Attention for Efficient Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11563","citing_title":"TCP-SSM: Efficient Vision State Space Models with Token-Conditioned Poles","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GM5CED6SAUEGHCZQTLULNMOGTS","json":"https://pith.science/pith/GM5CED6SAUEGHCZQTLULNMOGTS.json","graph_json":"https://pith.science/api/pith-number/GM5CED6SAUEGHCZQTLULNMOGTS/graph.json","events_json":"https://pith.science/api/pith-number/GM5CED6SAUEGHCZQTLULNMOGTS/events.json","paper":"https://pith.science/paper/GM5CED6S"},"agent_actions":{"view_html":"https://pith.science/pith/GM5CED6SAUEGHCZQTLULNMOGTS","download_json":"https://pith.science/pith/GM5CED6SAUEGHCZQTLULNMOGTS.json","view_paper":"https://pith.science/paper/GM5CED6S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.14052&json=true","fetch_graph":"https://pith.science/api/pith-number/GM5CED6SAUEGHCZQTLULNMOGTS/graph.json","fetch_events":"https://pith.science/api/pith-number/GM5CED6SAUEGHCZQTLULNMOGTS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GM5CED6SAUEGHCZQTLULNMOGTS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GM5CED6SAUEGHCZQTLULNMOGTS/action/storage_attestation","attest_author":"https://pith.science/pith/GM5CED6SAUEGHCZQTLULNMOGTS/action/author_attestation","sign_citation":"https://pith.science/pith/GM5CED6SAUEGHCZQTLULNMOGTS/action/citation_signature","submit_replication":"https://pith.science/pith/GM5CED6SAUEGHCZQTLULNMOGTS/action/replication_record"}},"created_at":"2026-07-05T06:05:32.688678+00:00","updated_at":"2026-07-05T06:05:32.688678+00:00"}