{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K4HR227YVURJOW7POYILPLXJLO","short_pith_number":"pith:K4HR227Y","schema_version":"1.0","canonical_sha256":"570f1d6bf8ad22975bef7610b7aee95bb4adc5a73175ba695b9a585b122ceed1","source":{"kind":"arxiv","id":"2405.16712","version":1},"attestation_state":"computed","paper":{"title":"Zamba: A Compact 7B SSM Hybrid Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Adam Ibrahim, Beren Millidge, James Whittington, Jonathan Pilault, Paolo Glorioso, Quentin Anthony, Yury Tokpanov","submitted_at":"2024-05-26T22:23:02Z","abstract_excerpt":"In this technical report, we present Zamba, a novel 7B SSM-transformer hybrid model which achieves competitive performance against leading open-weight models at a comparable scale. Zamba is trained on 1T tokens from openly available datasets and is the best non-transformer model at this scale. Zamba pioneers a unique architecture combining a Mamba backbone with a single shared attention module, thus obtaining the benefits of attention at minimal parameter cost. Due to its architecture, Zamba is significantly faster at inference than comparable transformer models and requires substantially less"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.16712","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-26T22:23:02Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"10402e7c3b8318e16b9aab7ef77e92a298d87dbcc9984c6afcfa9e75286896f3","abstract_canon_sha256":"ddc6862c0d840afd63dda8cef8c05a2128b09827ededaf2e6231aec0d2f58eec"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:23:34.371785Z","signature_b64":"EnJQqNCdQl4NIRPUdcMqTvCSrptP5YNnKjdOFe4s7GXiqu9tP7tU3QvskcGlWQdZGRe+nLq+/eLiyxsjuXvjAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"570f1d6bf8ad22975bef7610b7aee95bb4adc5a73175ba695b9a585b122ceed1","last_reissued_at":"2026-07-05T08:23:34.371322Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:23:34.371322Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Zamba: A Compact 7B SSM Hybrid Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Adam Ibrahim, Beren Millidge, James Whittington, Jonathan Pilault, Paolo Glorioso, Quentin Anthony, Yury Tokpanov","submitted_at":"2024-05-26T22:23:02Z","abstract_excerpt":"In this technical report, we present Zamba, a novel 7B SSM-transformer hybrid model which achieves competitive performance against leading open-weight models at a comparable scale. Zamba is trained on 1T tokens from openly available datasets and is the best non-transformer model at this scale. Zamba pioneers a unique architecture combining a Mamba backbone with a single shared attention module, thus obtaining the benefits of attention at minimal parameter cost. Due to its architecture, Zamba is significantly faster at inference than comparable transformer models and requires substantially less"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.16712","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.16712/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.16712","created_at":"2026-07-05T08:23:34.371379+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.16712v1","created_at":"2026-07-05T08:23:34.371379+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.16712","created_at":"2026-07-05T08:23:34.371379+00:00"},{"alias_kind":"pith_short_12","alias_value":"K4HR227YVURJ","created_at":"2026-07-05T08:23:34.371379+00:00"},{"alias_kind":"pith_short_16","alias_value":"K4HR227YVURJOW7P","created_at":"2026-07-05T08:23:34.371379+00:00"},{"alias_kind":"pith_short_8","alias_value":"K4HR227Y","created_at":"2026-07-05T08:23:34.371379+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":27,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24320","citing_title":"ZONOS2 Technical Report","ref_index":139,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02332","citing_title":"Forget Attention: Importance-Aware Attention Is All You Need","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00390","citing_title":"Zamba2-VL Technical Report","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24320","citing_title":"ZONOS2 Technical Report","ref_index":139,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30562","citing_title":"Morphing into Hybrid Attention Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03014","citing_title":"MOSAIC: Efficient Mixture-of-Agent Scheduling via Adaptive Aggregation and Inference Concurrency","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12364","citing_title":"On Subquadratic Architectures: From Applications to Principles","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2502.17873","citing_title":"An Efficient Self-Supervised Framework for Long-Sequence EEG Modeling","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2503.18970","citing_title":"Advancing Intelligent Sequence Modeling: Evolution, Trade-offs, and Applications of State- Space Architectures from S4 to Mamba","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21016","citing_title":"Gated KalmaNet: A Fading Memory Layer Through Test-Time Ridge Regression","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18826","citing_title":"The Routing and Filtering Structure of Attention","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2509.05276","citing_title":"SpikingBrain: Spiking Brain-inspired Large Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2406.07887","citing_title":"An Empirical Study of Mamba-based Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11794","citing_title":"DataComp-LM: In search of the next generation of training sets for language models","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2601.01972","citing_title":"Hidden State Poisoning Attacks against Mamba-based Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13215","citing_title":"When to Think Fast and Slow? AMOR: Adaptive Entropy Gate for Hybrid Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2506.13585","citing_title":"MiniMax-M1: Scaling Test-Time Compute Efficiently with Lightning Attention","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08301","citing_title":"Priming: Hybrid State Space Models From Pre-trained Transformers","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09516","citing_title":"Mixture of Layers with Hybrid Attention","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24715","citing_title":"Long-Context Aware Upcycling: A New Frontier for Hybrid LLM Scaling","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22442","citing_title":"HubRouter: A Pluggable Sub-Quadratic Routing Primitive for Hybrid Sequence Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05365","citing_title":"ZAYA1-8B Technical Report","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2405.21060","citing_title":"Transformers are SSMs: Generalized Models and Efficient Algorithms Through Structured State Space Duality","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07935","citing_title":"The Hyperscale Lottery: How State-Space Models Have Sacrificed Edge Efficiency","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07182","citing_title":"Star Elastic: Many-in-One Reasoning LLMs with Efficient Budget Control","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K4HR227YVURJOW7POYILPLXJLO","json":"https://pith.science/pith/K4HR227YVURJOW7POYILPLXJLO.json","graph_json":"https://pith.science/api/pith-number/K4HR227YVURJOW7POYILPLXJLO/graph.json","events_json":"https://pith.science/api/pith-number/K4HR227YVURJOW7POYILPLXJLO/events.json","paper":"https://pith.science/paper/K4HR227Y"},"agent_actions":{"view_html":"https://pith.science/pith/K4HR227YVURJOW7POYILPLXJLO","download_json":"https://pith.science/pith/K4HR227YVURJOW7POYILPLXJLO.json","view_paper":"https://pith.science/paper/K4HR227Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.16712&json=true","fetch_graph":"https://pith.science/api/pith-number/K4HR227YVURJOW7POYILPLXJLO/graph.json","fetch_events":"https://pith.science/api/pith-number/K4HR227YVURJOW7POYILPLXJLO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K4HR227YVURJOW7POYILPLXJLO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K4HR227YVURJOW7POYILPLXJLO/action/storage_attestation","attest_author":"https://pith.science/pith/K4HR227YVURJOW7POYILPLXJLO/action/author_attestation","sign_citation":"https://pith.science/pith/K4HR227YVURJOW7POYILPLXJLO/action/citation_signature","submit_replication":"https://pith.science/pith/K4HR227YVURJOW7POYILPLXJLO/action/replication_record"}},"created_at":"2026-07-05T08:23:34.371379+00:00","updated_at":"2026-07-05T08:23:34.371379+00:00"}