{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HAYLWA2YT7ZDRD3WZRSJFYZZDD","short_pith_number":"pith:HAYLWA2Y","schema_version":"1.0","canonical_sha256":"3830bb03589ff2388f76cc6492e33918d8c3d41e818a7bdb285d2a9856fb8a8f","source":{"kind":"arxiv","id":"2401.13660","version":3},"attestation_state":"computed","paper":{"title":"MambaByte: Token-free Selective State Space Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexander M. Rush, Jing Nathan Yan, Junxiong Wang, Tushaar Gangavarapu","submitted_at":"2024-01-24T18:53:53Z","abstract_excerpt":"Token-free language models learn directly from raw bytes and remove the inductive bias of subword tokenization. Operating on bytes, however, results in significantly longer sequences. In this setting, standard autoregressive Transformers scale poorly as the effective memory required grows with sequence length. The recent development of the Mamba state space model (SSM) offers an appealing alternative approach with a fixed-sized memory state and efficient decoding. We propose MambaByte, a token-free adaptation of the Mamba SSM trained autoregressively on byte sequences. In terms of modeling, we"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.13660","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-24T18:53:53Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"53821ed0335df6413bad5ec1a7cc72f6d02b4c72f99efd3b3c2683a4a53a88d5","abstract_canon_sha256":"4d93085b94e5540b07cf64127284c0c0ad18b68cc3eca07ab30e3117bbe4ec8c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:54:06.570194Z","signature_b64":"qGiu2ond7P/Ro1q9iSVaRmHAJRrzzZCBAzvMFLZUkG49HhJ/s5EZ5nZud7Weob3vEGjwSp2/k1NGmYT50VB0CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3830bb03589ff2388f76cc6492e33918d8c3d41e818a7bdb285d2a9856fb8a8f","last_reissued_at":"2026-07-05T08:54:06.569734Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:54:06.569734Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MambaByte: Token-free Selective State Space Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexander M. Rush, Jing Nathan Yan, Junxiong Wang, Tushaar Gangavarapu","submitted_at":"2024-01-24T18:53:53Z","abstract_excerpt":"Token-free language models learn directly from raw bytes and remove the inductive bias of subword tokenization. Operating on bytes, however, results in significantly longer sequences. In this setting, standard autoregressive Transformers scale poorly as the effective memory required grows with sequence length. The recent development of the Mamba state space model (SSM) offers an appealing alternative approach with a fixed-sized memory state and efficient decoding. We propose MambaByte, a token-free adaptation of the Mamba SSM trained autoregressively on byte sequences. In terms of modeling, we"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.13660","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.13660/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.13660","created_at":"2026-07-05T08:54:06.569791+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.13660v3","created_at":"2026-07-05T08:54:06.569791+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.13660","created_at":"2026-07-05T08:54:06.569791+00:00"},{"alias_kind":"pith_short_12","alias_value":"HAYLWA2YT7ZD","created_at":"2026-07-05T08:54:06.569791+00:00"},{"alias_kind":"pith_short_16","alias_value":"HAYLWA2YT7ZDRD3W","created_at":"2026-07-05T08:54:06.569791+00:00"},{"alias_kind":"pith_short_8","alias_value":"HAYLWA2Y","created_at":"2026-07-05T08:54:06.569791+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11642","citing_title":"3-Key-Input: Exploring the Theoretical Minimum Keys for Text Entry","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04552","citing_title":"LDARNet: DNA Adaptive Representation Network with Learnable Tokenization for Genomic Modeling","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2404.07106","citing_title":"3DMambaComplete: Exploring Structured State Space Model for Point Cloud Completion","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2408.01129","citing_title":"A Survey of Mamba","ref_index":190,"is_internal_anchor":false},{"citing_arxiv_id":"2402.19427","citing_title":"Griffin: Mixing Gated Linear Recurrences with Local Attention for Efficient Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2312.06635","citing_title":"Gated Linear Attention Transformers with Hardware-Efficient Training","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12928","citing_title":"The Efficiency Gap in Byte Modeling","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11034","citing_title":"MambaNetBurst: Direct Byte-level Network Traffic Classification without Tokenization or Pretraining","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09630","citing_title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","ref_index":90,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HAYLWA2YT7ZDRD3WZRSJFYZZDD","json":"https://pith.science/pith/HAYLWA2YT7ZDRD3WZRSJFYZZDD.json","graph_json":"https://pith.science/api/pith-number/HAYLWA2YT7ZDRD3WZRSJFYZZDD/graph.json","events_json":"https://pith.science/api/pith-number/HAYLWA2YT7ZDRD3WZRSJFYZZDD/events.json","paper":"https://pith.science/paper/HAYLWA2Y"},"agent_actions":{"view_html":"https://pith.science/pith/HAYLWA2YT7ZDRD3WZRSJFYZZDD","download_json":"https://pith.science/pith/HAYLWA2YT7ZDRD3WZRSJFYZZDD.json","view_paper":"https://pith.science/paper/HAYLWA2Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.13660&json=true","fetch_graph":"https://pith.science/api/pith-number/HAYLWA2YT7ZDRD3WZRSJFYZZDD/graph.json","fetch_events":"https://pith.science/api/pith-number/HAYLWA2YT7ZDRD3WZRSJFYZZDD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HAYLWA2YT7ZDRD3WZRSJFYZZDD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HAYLWA2YT7ZDRD3WZRSJFYZZDD/action/storage_attestation","attest_author":"https://pith.science/pith/HAYLWA2YT7ZDRD3WZRSJFYZZDD/action/author_attestation","sign_citation":"https://pith.science/pith/HAYLWA2YT7ZDRD3WZRSJFYZZDD/action/citation_signature","submit_replication":"https://pith.science/pith/HAYLWA2YT7ZDRD3WZRSJFYZZDD/action/replication_record"}},"created_at":"2026-07-05T08:54:06.569791+00:00","updated_at":"2026-07-05T08:54:06.569791+00:00"}