{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:6FD732FFOUT7XZLSAQ23FJHQAE","short_pith_number":"pith:6FD732FF","schema_version":"1.0","canonical_sha256":"f147fde8a57527fbe5720435b2a4f0011e3d3cf4d3dd27fa64b2715d90eb0499","source":{"kind":"arxiv","id":"2206.13947","version":3},"attestation_state":"computed","paper":{"title":"Long Range Language Modeling via Gated State Spaces","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Ankit Gupta, Ashok Cutkosky, Behnam Neyshabur, Harsh Mehta","submitted_at":"2022-06-27T01:50:18Z","abstract_excerpt":"State space models have shown to be effective at modeling long range dependencies, specially on sequence classification tasks. In this work we focus on autoregressive sequence modeling over English books, Github source code and ArXiv mathematics articles. Based on recent developments around the effectiveness of gated activation functions, we propose a new layer named Gated State Space (GSS) and show that it trains significantly faster than the diagonal version of S4 (i.e. DSS) on TPUs, is fairly competitive with several well-tuned Transformer-based baselines and exhibits zero-shot generalizati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.13947","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-06-27T01:50:18Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"c7c972e5fca0e273daf5cbd54c0a0a09a6aa34efc080e98d43675ab1d22f850b","abstract_canon_sha256":"02ac760aa52a444881a2f4992c25475bab7f8bc2fbc82169773fa7f0b6289544"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:36:49.998836Z","signature_b64":"Op7jslIAzajxZaNgKxA08q08BJwKsqwmF2Hv9TR4H+dCkt/el3AL/IxEzszW33t/628vKQ5T2KqCG2kh0RoKBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f147fde8a57527fbe5720435b2a4f0011e3d3cf4d3dd27fa64b2715d90eb0499","last_reissued_at":"2026-07-05T04:36:49.998389Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:36:49.998389Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Long Range Language Modeling via Gated State Spaces","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Ankit Gupta, Ashok Cutkosky, Behnam Neyshabur, Harsh Mehta","submitted_at":"2022-06-27T01:50:18Z","abstract_excerpt":"State space models have shown to be effective at modeling long range dependencies, specially on sequence classification tasks. In this work we focus on autoregressive sequence modeling over English books, Github source code and ArXiv mathematics articles. Based on recent developments around the effectiveness of gated activation functions, we propose a new layer named Gated State Space (GSS) and show that it trains significantly faster than the diagonal version of S4 (i.e. DSS) on TPUs, is fairly competitive with several well-tuned Transformer-based baselines and exhibits zero-shot generalizati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.13947","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.13947/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.13947","created_at":"2026-07-05T04:36:49.998446+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.13947v3","created_at":"2026-07-05T04:36:49.998446+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.13947","created_at":"2026-07-05T04:36:49.998446+00:00"},{"alias_kind":"pith_short_12","alias_value":"6FD732FFOUT7","created_at":"2026-07-05T04:36:49.998446+00:00"},{"alias_kind":"pith_short_16","alias_value":"6FD732FFOUT7XZLS","created_at":"2026-07-05T04:36:49.998446+00:00"},{"alias_kind":"pith_short_8","alias_value":"6FD732FF","created_at":"2026-07-05T04:36:49.998446+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19932","citing_title":"Spatial-Aware Reduction Framework: Towards Efficient and Faithful Visual State Space Models","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12895","citing_title":"LongSpike: Fractional Order Spiking State Space Models for Efficient Long Sequence Learning","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09432","citing_title":"Graph Mamba Operator: A Latent Simulator for Interacting Particle Systems","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2404.07106","citing_title":"3DMambaComplete: Exploring Structured State Space Model for Point Cloud Completion","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2503.18970","citing_title":"Advancing Intelligent Sequence Modeling: Evolution, Trade-offs, and Applications of State- Space Architectures from S4 to Mamba","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2402.19427","citing_title":"Griffin: Mixing Gated Linear Recurrences with Local Attention for Efficient Language Models","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6FD732FFOUT7XZLSAQ23FJHQAE","json":"https://pith.science/pith/6FD732FFOUT7XZLSAQ23FJHQAE.json","graph_json":"https://pith.science/api/pith-number/6FD732FFOUT7XZLSAQ23FJHQAE/graph.json","events_json":"https://pith.science/api/pith-number/6FD732FFOUT7XZLSAQ23FJHQAE/events.json","paper":"https://pith.science/paper/6FD732FF"},"agent_actions":{"view_html":"https://pith.science/pith/6FD732FFOUT7XZLSAQ23FJHQAE","download_json":"https://pith.science/pith/6FD732FFOUT7XZLSAQ23FJHQAE.json","view_paper":"https://pith.science/paper/6FD732FF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.13947&json=true","fetch_graph":"https://pith.science/api/pith-number/6FD732FFOUT7XZLSAQ23FJHQAE/graph.json","fetch_events":"https://pith.science/api/pith-number/6FD732FFOUT7XZLSAQ23FJHQAE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6FD732FFOUT7XZLSAQ23FJHQAE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6FD732FFOUT7XZLSAQ23FJHQAE/action/storage_attestation","attest_author":"https://pith.science/pith/6FD732FFOUT7XZLSAQ23FJHQAE/action/author_attestation","sign_citation":"https://pith.science/pith/6FD732FFOUT7XZLSAQ23FJHQAE/action/citation_signature","submit_replication":"https://pith.science/pith/6FD732FFOUT7XZLSAQ23FJHQAE/action/replication_record"}},"created_at":"2026-07-05T04:36:49.998446+00:00","updated_at":"2026-07-05T04:36:49.998446+00:00"}