{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZSNPJWJJJNNVTRWPI2OOJSKH4B","short_pith_number":"pith:ZSNPJWJJ","schema_version":"1.0","canonical_sha256":"cc9af4d9294b5b59c6cf469ce4c947e055f5672b81b20a4b9e44b5627aeb536b","source":{"kind":"arxiv","id":"2405.18832","version":1},"attestation_state":"computed","paper":{"title":"MoNDE: Mixture of Near-Data Experts for Large-Scale Sparse Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.AR"],"primary_cat":"cs.LG","authors_text":"Hyuk-jae Lee, Jaehoon Cho, Jaewoong Sim, Kwanseok Choi, Taehyun Kim, Youngmock Cho","submitted_at":"2024-05-29T07:23:29Z","abstract_excerpt":"Mixture-of-Experts (MoE) large language models (LLM) have memory requirements that often exceed the GPU memory capacity, requiring costly parameter movement from secondary memories to the GPU for expert computation. In this work, we present Mixture of Near-Data Experts (MoNDE), a near-data computing solution that efficiently enables MoE LLM inference. MoNDE reduces the volume of MoE parameter movement by transferring only the $\\textit{hot}$ experts to the GPU, while computing the remaining $\\textit{cold}$ experts inside the host memory device. By replacing the transfers of massive expert param"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.18832","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-29T07:23:29Z","cross_cats_sorted":["cs.AI","cs.AR"],"title_canon_sha256":"f28ba94acce5e86323e9c148a8fb43f421df13b77b22e14929fe1ca0d4103c03","abstract_canon_sha256":"a1af0a780e3967babff0016117c6a6d36f60a950455312fcb22d3aaf93a39582"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:24:43.921054Z","signature_b64":"yuBEmaGpzjw8sTX73EBMLzjx9l+4zn8zzQ93W2JXi0iNO6Pvw0HZa+OHINT1YwoxoBGACkhC+EUhCW87T2eEBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc9af4d9294b5b59c6cf469ce4c947e055f5672b81b20a4b9e44b5627aeb536b","last_reissued_at":"2026-07-05T08:24:43.920588Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:24:43.920588Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MoNDE: Mixture of Near-Data Experts for Large-Scale Sparse Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.AR"],"primary_cat":"cs.LG","authors_text":"Hyuk-jae Lee, Jaehoon Cho, Jaewoong Sim, Kwanseok Choi, Taehyun Kim, Youngmock Cho","submitted_at":"2024-05-29T07:23:29Z","abstract_excerpt":"Mixture-of-Experts (MoE) large language models (LLM) have memory requirements that often exceed the GPU memory capacity, requiring costly parameter movement from secondary memories to the GPU for expert computation. In this work, we present Mixture of Near-Data Experts (MoNDE), a near-data computing solution that efficiently enables MoE LLM inference. MoNDE reduces the volume of MoE parameter movement by transferring only the $\\textit{hot}$ experts to the GPU, while computing the remaining $\\textit{cold}$ experts inside the host memory device. By replacing the transfers of massive expert param"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.18832","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.18832/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.18832","created_at":"2026-07-05T08:24:43.920645+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.18832v1","created_at":"2026-07-05T08:24:43.920645+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.18832","created_at":"2026-07-05T08:24:43.920645+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZSNPJWJJJNNV","created_at":"2026-07-05T08:24:43.920645+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZSNPJWJJJNNVTRWP","created_at":"2026-07-05T08:24:43.920645+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZSNPJWJJ","created_at":"2026-07-05T08:24:43.920645+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.05639","citing_title":"TokenStack: A Heterogeneous HBM-PIM Architecture and Runtime for Efficient LLM Inference","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZSNPJWJJJNNVTRWPI2OOJSKH4B","json":"https://pith.science/pith/ZSNPJWJJJNNVTRWPI2OOJSKH4B.json","graph_json":"https://pith.science/api/pith-number/ZSNPJWJJJNNVTRWPI2OOJSKH4B/graph.json","events_json":"https://pith.science/api/pith-number/ZSNPJWJJJNNVTRWPI2OOJSKH4B/events.json","paper":"https://pith.science/paper/ZSNPJWJJ"},"agent_actions":{"view_html":"https://pith.science/pith/ZSNPJWJJJNNVTRWPI2OOJSKH4B","download_json":"https://pith.science/pith/ZSNPJWJJJNNVTRWPI2OOJSKH4B.json","view_paper":"https://pith.science/paper/ZSNPJWJJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.18832&json=true","fetch_graph":"https://pith.science/api/pith-number/ZSNPJWJJJNNVTRWPI2OOJSKH4B/graph.json","fetch_events":"https://pith.science/api/pith-number/ZSNPJWJJJNNVTRWPI2OOJSKH4B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZSNPJWJJJNNVTRWPI2OOJSKH4B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZSNPJWJJJNNVTRWPI2OOJSKH4B/action/storage_attestation","attest_author":"https://pith.science/pith/ZSNPJWJJJNNVTRWPI2OOJSKH4B/action/author_attestation","sign_citation":"https://pith.science/pith/ZSNPJWJJJNNVTRWPI2OOJSKH4B/action/citation_signature","submit_replication":"https://pith.science/pith/ZSNPJWJJJNNVTRWPI2OOJSKH4B/action/replication_record"}},"created_at":"2026-07-05T08:24:43.920645+00:00","updated_at":"2026-07-05T08:24:43.920645+00:00"}