{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MJF5YUCTSDWB4NY6LN4V6MWQN3","short_pith_number":"pith:MJF5YUCT","schema_version":"1.0","canonical_sha256":"624bdc505390ec1e371e5b795f32d06ed3c99987fbf20da47b17cb8050510a47","source":{"kind":"arxiv","id":"2408.12570","version":1},"attestation_state":"computed","paper":{"title":"Jamba-1.5: Hybrid Transformer-Mamba Models at Scale","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alan Arazi, Amir Bergman, Avshalom Manevich, Barak Peleg, Ben Aviram, Chen Almagor, Clara Fridman, Daniel Gissin, Daniel Jannai, Dan Padnos, Dor Muhlgay, Dor Zimberg, Edden M Gerber, Elad Dolev, Eran Krakovsky, Erez Safahi, Erez Schwartz, Gal Cohen, Gal Shachaf, Haim Rozenblum, Hofit Bata, Ido Blass, Inbal Magar, Itay Dalmedigos, Jamba Team: Barak Lenz, Jhonathan Osin, Julie Fadlon, Maria Rozman, Matan Danos, Michael Gokhman, Mor Zusman, Naama Gidron, Nir Ratner, Noam Gat, Noam Rozen, Oded Fried, Ohad Leshno, Omer Antverg, Omri Abend, Opher Lieber, Or Dagan, Orit Cohavi, Raz Alon, Ro'i Belson, Roi Cohen, Roman Glozman, Rom Gilad, Shahar Lev, Shaked Meirom, Tal Delbari, Tal Ness, Tom Ben Gal, Tom Braude, Tomer Asida, Uriya Pumerantz, Yehoshua Cohen, Yoav Shoham, Yonatan Belinkov, Yuval Globerson, Yuval Peleg Levy","submitted_at":"2024-08-22T17:38:59Z","abstract_excerpt":"We present Jamba-1.5, new instruction-tuned large language models based on our Jamba architecture. Jamba is a hybrid Transformer-Mamba mixture of experts architecture, providing high throughput and low memory usage across context lengths, while retaining the same or better quality as Transformer models. We release two model sizes: Jamba-1.5-Large, with 94B active parameters, and Jamba-1.5-Mini, with 12B active parameters. Both models are fine-tuned for a variety of conversational and instruction-following capabilties, and have an effective context length of 256K tokens, the largest amongst ope"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.12570","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-08-22T17:38:59Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a93a7af23db130e676b0067cf7633e1c21182278faa81d7a4130217f7979ac79","abstract_canon_sha256":"3e676cb24698abd7bf5bde0d64adf47ac315761ecb86d88679763ce2180ff709"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:58:16.794843Z","signature_b64":"8pvy6VfCgMm6Hc49pCyPiFh4PE13T7VA5TcFXfYs2sueWxMAHNr6iEyFUiFcuXFi+le64kkfQEylHFQQwzOKBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"624bdc505390ec1e371e5b795f32d06ed3c99987fbf20da47b17cb8050510a47","last_reissued_at":"2026-07-05T08:58:16.794247Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:58:16.794247Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Jamba-1.5: Hybrid Transformer-Mamba Models at Scale","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alan Arazi, Amir Bergman, Avshalom Manevich, Barak Peleg, Ben Aviram, Chen Almagor, Clara Fridman, Daniel Gissin, Daniel Jannai, Dan Padnos, Dor Muhlgay, Dor Zimberg, Edden M Gerber, Elad Dolev, Eran Krakovsky, Erez Safahi, Erez Schwartz, Gal Cohen, Gal Shachaf, Haim Rozenblum, Hofit Bata, Ido Blass, Inbal Magar, Itay Dalmedigos, Jamba Team: Barak Lenz, Jhonathan Osin, Julie Fadlon, Maria Rozman, Matan Danos, Michael Gokhman, Mor Zusman, Naama Gidron, Nir Ratner, Noam Gat, Noam Rozen, Oded Fried, Ohad Leshno, Omer Antverg, Omri Abend, Opher Lieber, Or Dagan, Orit Cohavi, Raz Alon, Ro'i Belson, Roi Cohen, Roman Glozman, Rom Gilad, Shahar Lev, Shaked Meirom, Tal Delbari, Tal Ness, Tom Ben Gal, Tom Braude, Tomer Asida, Uriya Pumerantz, Yehoshua Cohen, Yoav Shoham, Yonatan Belinkov, Yuval Globerson, Yuval Peleg Levy","submitted_at":"2024-08-22T17:38:59Z","abstract_excerpt":"We present Jamba-1.5, new instruction-tuned large language models based on our Jamba architecture. Jamba is a hybrid Transformer-Mamba mixture of experts architecture, providing high throughput and low memory usage across context lengths, while retaining the same or better quality as Transformer models. We release two model sizes: Jamba-1.5-Large, with 94B active parameters, and Jamba-1.5-Mini, with 12B active parameters. Both models are fine-tuned for a variety of conversational and instruction-following capabilties, and have an effective context length of 256K tokens, the largest amongst ope"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.12570","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.12570/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.12570","created_at":"2026-07-05T08:58:16.794333+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.12570v1","created_at":"2026-07-05T08:58:16.794333+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.12570","created_at":"2026-07-05T08:58:16.794333+00:00"},{"alias_kind":"pith_short_12","alias_value":"MJF5YUCTSDWB","created_at":"2026-07-05T08:58:16.794333+00:00"},{"alias_kind":"pith_short_16","alias_value":"MJF5YUCTSDWB4NY6","created_at":"2026-07-05T08:58:16.794333+00:00"},{"alias_kind":"pith_short_8","alias_value":"MJF5YUCT","created_at":"2026-07-05T08:58:16.794333+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21075","citing_title":"FiLM-Coordinated Dual-Branch Transformer for Global-Local Dependency Modeling in Language Modeling","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03014","citing_title":"MOSAIC: Efficient Mixture-of-Agent Scheduling via Adaptive Aggregation and Inference Concurrency","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2410.13846","citing_title":"LightTransfer: Your Long-Context LLM is Secretly a Hybrid Model with Effortless Adaptation","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2503.18970","citing_title":"Advancing Intelligent Sequence Modeling: Evolution, Trade-offs, and Applications of State-Space Architectures from S4 to Mamba","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04800","citing_title":"Hybrid Architectures for Language Models: Systematic Analysis and Design Insights","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26083","citing_title":"Nirvana: A Specialized Generalist Model With Task-Aware Memory Mechanism","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2601.01972","citing_title":"Hidden State Poisoning Attacks against Mamba-based Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2410.10781","citing_title":"When Attention Sink Emerges in Language Models: An Empirical View","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2506.13585","citing_title":"MiniMax-M1: Scaling Test-Time Compute Efficiently with Lightning Attention","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MJF5YUCTSDWB4NY6LN4V6MWQN3","json":"https://pith.science/pith/MJF5YUCTSDWB4NY6LN4V6MWQN3.json","graph_json":"https://pith.science/api/pith-number/MJF5YUCTSDWB4NY6LN4V6MWQN3/graph.json","events_json":"https://pith.science/api/pith-number/MJF5YUCTSDWB4NY6LN4V6MWQN3/events.json","paper":"https://pith.science/paper/MJF5YUCT"},"agent_actions":{"view_html":"https://pith.science/pith/MJF5YUCTSDWB4NY6LN4V6MWQN3","download_json":"https://pith.science/pith/MJF5YUCTSDWB4NY6LN4V6MWQN3.json","view_paper":"https://pith.science/paper/MJF5YUCT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.12570&json=true","fetch_graph":"https://pith.science/api/pith-number/MJF5YUCTSDWB4NY6LN4V6MWQN3/graph.json","fetch_events":"https://pith.science/api/pith-number/MJF5YUCTSDWB4NY6LN4V6MWQN3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MJF5YUCTSDWB4NY6LN4V6MWQN3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MJF5YUCTSDWB4NY6LN4V6MWQN3/action/storage_attestation","attest_author":"https://pith.science/pith/MJF5YUCTSDWB4NY6LN4V6MWQN3/action/author_attestation","sign_citation":"https://pith.science/pith/MJF5YUCTSDWB4NY6LN4V6MWQN3/action/citation_signature","submit_replication":"https://pith.science/pith/MJF5YUCTSDWB4NY6LN4V6MWQN3/action/replication_record"}},"created_at":"2026-07-05T08:58:16.794333+00:00","updated_at":"2026-07-05T08:58:16.794333+00:00"}