{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JUCIS72WKBCDCTJBVDQJAD5OLI","short_pith_number":"pith:JUCIS72W","schema_version":"1.0","canonical_sha256":"4d04897f565044314d21a8e0900fae5a3fbf5ca6f2c34af26c0f3269dba0ebc1","source":{"kind":"arxiv","id":"2509.02408","version":1},"attestation_state":"computed","paper":{"title":"Cache Management for Mixture-of-Experts LLMs -- extended version","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DS"],"primary_cat":"cs.LG","authors_text":"Adrien Obrecht, Bertrand Simon, Loris Marchal, Spyros Angelopoulos","submitted_at":"2025-09-02T15:19:06Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable capabilities across a variety of tasks. One of the main challenges towards the successful deployment of LLMs is memory management, since they typically involve billions of parameters. To this end, architectures based on Mixture-of-Experts have been proposed, which aim to reduce the size of the parameters that are activated when producing a token. This raises the equally critical issue of efficiently managing the limited cache of the system, in that frequently used experts should be stored in the fast cache rather than in the slower seco"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.02408","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-09-02T15:19:06Z","cross_cats_sorted":["cs.DS"],"title_canon_sha256":"a05af24d8cdfc30e71cae2d8e5e317ead9ff46f3e4ddfdb1e36b6d29bc4d68a3","abstract_canon_sha256":"9b164bb1e3a0d500ed1bd26db997a0361c93ea3db50a0a110888090d54f9df9a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:03:38.017489Z","signature_b64":"DeBEnJMErUobIa9+/UsD7YdRn2vTldNEQNMXYM9Bk0omWD0IdCruHERNdxogzLuFix5phnQisf+fbR2a9nKfCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4d04897f565044314d21a8e0900fae5a3fbf5ca6f2c34af26c0f3269dba0ebc1","last_reissued_at":"2026-07-05T12:03:38.016998Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:03:38.016998Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cache Management for Mixture-of-Experts LLMs -- extended version","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DS"],"primary_cat":"cs.LG","authors_text":"Adrien Obrecht, Bertrand Simon, Loris Marchal, Spyros Angelopoulos","submitted_at":"2025-09-02T15:19:06Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable capabilities across a variety of tasks. One of the main challenges towards the successful deployment of LLMs is memory management, since they typically involve billions of parameters. To this end, architectures based on Mixture-of-Experts have been proposed, which aim to reduce the size of the parameters that are activated when producing a token. This raises the equally critical issue of efficiently managing the limited cache of the system, in that frequently used experts should be stored in the fast cache rather than in the slower seco"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.02408","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.02408/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.02408","created_at":"2026-07-05T12:03:38.017062+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.02408v1","created_at":"2026-07-05T12:03:38.017062+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.02408","created_at":"2026-07-05T12:03:38.017062+00:00"},{"alias_kind":"pith_short_12","alias_value":"JUCIS72WKBCD","created_at":"2026-07-05T12:03:38.017062+00:00"},{"alias_kind":"pith_short_16","alias_value":"JUCIS72WKBCDCTJB","created_at":"2026-07-05T12:03:38.017062+00:00"},{"alias_kind":"pith_short_8","alias_value":"JUCIS72W","created_at":"2026-07-05T12:03:38.017062+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JUCIS72WKBCDCTJBVDQJAD5OLI","json":"https://pith.science/pith/JUCIS72WKBCDCTJBVDQJAD5OLI.json","graph_json":"https://pith.science/api/pith-number/JUCIS72WKBCDCTJBVDQJAD5OLI/graph.json","events_json":"https://pith.science/api/pith-number/JUCIS72WKBCDCTJBVDQJAD5OLI/events.json","paper":"https://pith.science/paper/JUCIS72W"},"agent_actions":{"view_html":"https://pith.science/pith/JUCIS72WKBCDCTJBVDQJAD5OLI","download_json":"https://pith.science/pith/JUCIS72WKBCDCTJBVDQJAD5OLI.json","view_paper":"https://pith.science/paper/JUCIS72W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.02408&json=true","fetch_graph":"https://pith.science/api/pith-number/JUCIS72WKBCDCTJBVDQJAD5OLI/graph.json","fetch_events":"https://pith.science/api/pith-number/JUCIS72WKBCDCTJBVDQJAD5OLI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JUCIS72WKBCDCTJBVDQJAD5OLI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JUCIS72WKBCDCTJBVDQJAD5OLI/action/storage_attestation","attest_author":"https://pith.science/pith/JUCIS72WKBCDCTJBVDQJAD5OLI/action/author_attestation","sign_citation":"https://pith.science/pith/JUCIS72WKBCDCTJBVDQJAD5OLI/action/citation_signature","submit_replication":"https://pith.science/pith/JUCIS72WKBCDCTJBVDQJAD5OLI/action/replication_record"}},"created_at":"2026-07-05T12:03:38.017062+00:00","updated_at":"2026-07-05T12:03:38.017062+00:00"}