{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SKVJJIGL7GG3GGSC3DA22PGNCI","short_pith_number":"pith:SKVJJIGL","schema_version":"1.0","canonical_sha256":"92aa94a0cbf98db31a42d8c1ad3ccd120258cf283775aaec2ec90228ca32fdfe","source":{"kind":"arxiv","id":"2405.07518","version":2},"attestation_state":"computed","paper":{"title":"SambaNova SN40L: Scaling the AI Memory Wall with Dataflow and Composition of Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.AR","authors_text":"Angela Wang, Apurv Vivek, Arjun Sabnis, Calvin Leung, Darshan Gandhi, David Jackson, Dawei Huang, Denis Sokolov, Edison Chen, Jiayu Bai, Joshua Brot, Kaizhao Liang, Karen Li, Kejie Zhang, Kevin J. Brown, Kunle Olukotun, Manish K. Shah, Mark Gottscho, Mark Luttrell, Mingran Wang, Raghu Prabhakar, Ram Sivaramakrishnan, Sumti Jairath, Swayambhoo Jain, Tianren Gao, Tuowen Zhao, Urmish Thakker, Xiangyu Song, Yongning Sheng, Yun Du","submitted_at":"2024-05-13T07:32:45Z","abstract_excerpt":"Monolithic large language models (LLMs) like GPT-4 have paved the way for modern generative AI applications. Training, serving, and maintaining monolithic LLMs at scale, however, remains prohibitively expensive and challenging. The disproportionate increase in compute-to-memory ratio of modern AI accelerators have created a memory wall, necessitating new methods to deploy AI. Composition of Experts (CoE) is an alternative modular approach that lowers the cost and complexity of training and serving. However, this approach presents two key challenges when using conventional hardware: (1) without"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.07518","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AR","submitted_at":"2024-05-13T07:32:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3577481ce09fc24612221eef12f14c8b2423faa229205b0eb356bcb5a61690e0","abstract_canon_sha256":"613e0d40151e8246d3ac0732a6ec1b7aee82dba34be8afe495d84c34c932e41c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:31:02.309379Z","signature_b64":"tvk+UBWiaEEU1n5tsxVlaj8sA4HinjxsW1utTAwK6HBSXfip6iPKGk3qqv8IS6Qh896LxLCjdwE20OkH7CfABw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"92aa94a0cbf98db31a42d8c1ad3ccd120258cf283775aaec2ec90228ca32fdfe","last_reissued_at":"2026-07-05T09:31:02.308888Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:31:02.308888Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SambaNova SN40L: Scaling the AI Memory Wall with Dataflow and Composition of Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.AR","authors_text":"Angela Wang, Apurv Vivek, Arjun Sabnis, Calvin Leung, Darshan Gandhi, David Jackson, Dawei Huang, Denis Sokolov, Edison Chen, Jiayu Bai, Joshua Brot, Kaizhao Liang, Karen Li, Kejie Zhang, Kevin J. Brown, Kunle Olukotun, Manish K. Shah, Mark Gottscho, Mark Luttrell, Mingran Wang, Raghu Prabhakar, Ram Sivaramakrishnan, Sumti Jairath, Swayambhoo Jain, Tianren Gao, Tuowen Zhao, Urmish Thakker, Xiangyu Song, Yongning Sheng, Yun Du","submitted_at":"2024-05-13T07:32:45Z","abstract_excerpt":"Monolithic large language models (LLMs) like GPT-4 have paved the way for modern generative AI applications. Training, serving, and maintaining monolithic LLMs at scale, however, remains prohibitively expensive and challenging. The disproportionate increase in compute-to-memory ratio of modern AI accelerators have created a memory wall, necessitating new methods to deploy AI. Composition of Experts (CoE) is an alternative modular approach that lowers the cost and complexity of training and serving. However, this approach presents two key challenges when using conventional hardware: (1) without"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.07518","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.07518/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.07518","created_at":"2026-07-05T09:31:02.308947+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.07518v2","created_at":"2026-07-05T09:31:02.308947+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.07518","created_at":"2026-07-05T09:31:02.308947+00:00"},{"alias_kind":"pith_short_12","alias_value":"SKVJJIGL7GG3","created_at":"2026-07-05T09:31:02.308947+00:00"},{"alias_kind":"pith_short_16","alias_value":"SKVJJIGL7GG3GGSC","created_at":"2026-07-05T09:31:02.308947+00:00"},{"alias_kind":"pith_short_8","alias_value":"SKVJJIGL","created_at":"2026-07-05T09:31:02.308947+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.11506","citing_title":"ELK: Exploring the Efficiency of Inter-core Connected AI Chips with Deep Learning Compiler Techniques","ref_index":46,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SKVJJIGL7GG3GGSC3DA22PGNCI","json":"https://pith.science/pith/SKVJJIGL7GG3GGSC3DA22PGNCI.json","graph_json":"https://pith.science/api/pith-number/SKVJJIGL7GG3GGSC3DA22PGNCI/graph.json","events_json":"https://pith.science/api/pith-number/SKVJJIGL7GG3GGSC3DA22PGNCI/events.json","paper":"https://pith.science/paper/SKVJJIGL"},"agent_actions":{"view_html":"https://pith.science/pith/SKVJJIGL7GG3GGSC3DA22PGNCI","download_json":"https://pith.science/pith/SKVJJIGL7GG3GGSC3DA22PGNCI.json","view_paper":"https://pith.science/paper/SKVJJIGL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.07518&json=true","fetch_graph":"https://pith.science/api/pith-number/SKVJJIGL7GG3GGSC3DA22PGNCI/graph.json","fetch_events":"https://pith.science/api/pith-number/SKVJJIGL7GG3GGSC3DA22PGNCI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SKVJJIGL7GG3GGSC3DA22PGNCI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SKVJJIGL7GG3GGSC3DA22PGNCI/action/storage_attestation","attest_author":"https://pith.science/pith/SKVJJIGL7GG3GGSC3DA22PGNCI/action/author_attestation","sign_citation":"https://pith.science/pith/SKVJJIGL7GG3GGSC3DA22PGNCI/action/citation_signature","submit_replication":"https://pith.science/pith/SKVJJIGL7GG3GGSC3DA22PGNCI/action/replication_record"}},"created_at":"2026-07-05T09:31:02.308947+00:00","updated_at":"2026-07-05T09:31:02.308947+00:00"}