{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NUS4P5UED2TJTXLR6H2WXA62GE","short_pith_number":"pith:NUS4P5UE","schema_version":"1.0","canonical_sha256":"6d25c7f6841ea699dd71f1f56b83da31239e82ab720df453e8a2c943d7b53fbd","source":{"kind":"arxiv","id":"2501.10375","version":2},"attestation_state":"computed","paper":{"title":"DAOP: Data-Aware Offloading and Predictive Pre-Calculation for Efficient MoE Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Shivam Aggarwal, Tulika Mitra, Yujie Zhang","submitted_at":"2024-12-16T07:59:21Z","abstract_excerpt":"Mixture-of-Experts (MoE) models, though highly effective for various machine learning tasks, face significant deployment challenges on memory-constrained devices. While GPUs offer fast inference, their limited memory compared to CPUs means not all experts can be stored on the GPU simultaneously, necessitating frequent, costly data transfers from CPU memory, often negating GPU speed advantages. To address this, we present DAOP, an on-device MoE inference engine to optimize parallel GPU-CPU execution. DAOP dynamically allocates experts between CPU and GPU based on per-sequence activation pattern"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.10375","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2024-12-16T07:59:21Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"0d212268157673bf29bb234314705e7dc839e1cf55d723411b357cb7ee55751b","abstract_canon_sha256":"3d53bfc8769ac838a835c65c8b76c0f4bfda866a4116b594c73e56473095e9aa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:58:22.510870Z","signature_b64":"9Wvl2WpnQye3g3kRjSZ3+r+LSVHtkFWgVTQlq9FzgRvQ9ccYyJvAkHimWa5gd8q/T4WydgVU0XCwdC6RAOvXBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6d25c7f6841ea699dd71f1f56b83da31239e82ab720df453e8a2c943d7b53fbd","last_reissued_at":"2026-07-05T10:58:22.510353Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:58:22.510353Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DAOP: Data-Aware Offloading and Predictive Pre-Calculation for Efficient MoE Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Shivam Aggarwal, Tulika Mitra, Yujie Zhang","submitted_at":"2024-12-16T07:59:21Z","abstract_excerpt":"Mixture-of-Experts (MoE) models, though highly effective for various machine learning tasks, face significant deployment challenges on memory-constrained devices. While GPUs offer fast inference, their limited memory compared to CPUs means not all experts can be stored on the GPU simultaneously, necessitating frequent, costly data transfers from CPU memory, often negating GPU speed advantages. To address this, we present DAOP, an on-device MoE inference engine to optimize parallel GPU-CPU execution. DAOP dynamically allocates experts between CPU and GPU based on per-sequence activation pattern"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.10375","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.10375/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.10375","created_at":"2026-07-05T10:58:22.510419+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.10375v2","created_at":"2026-07-05T10:58:22.510419+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.10375","created_at":"2026-07-05T10:58:22.510419+00:00"},{"alias_kind":"pith_short_12","alias_value":"NUS4P5UED2TJ","created_at":"2026-07-05T10:58:22.510419+00:00"},{"alias_kind":"pith_short_16","alias_value":"NUS4P5UED2TJTXLR","created_at":"2026-07-05T10:58:22.510419+00:00"},{"alias_kind":"pith_short_8","alias_value":"NUS4P5UE","created_at":"2026-07-05T10:58:22.510419+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NUS4P5UED2TJTXLR6H2WXA62GE","json":"https://pith.science/pith/NUS4P5UED2TJTXLR6H2WXA62GE.json","graph_json":"https://pith.science/api/pith-number/NUS4P5UED2TJTXLR6H2WXA62GE/graph.json","events_json":"https://pith.science/api/pith-number/NUS4P5UED2TJTXLR6H2WXA62GE/events.json","paper":"https://pith.science/paper/NUS4P5UE"},"agent_actions":{"view_html":"https://pith.science/pith/NUS4P5UED2TJTXLR6H2WXA62GE","download_json":"https://pith.science/pith/NUS4P5UED2TJTXLR6H2WXA62GE.json","view_paper":"https://pith.science/paper/NUS4P5UE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.10375&json=true","fetch_graph":"https://pith.science/api/pith-number/NUS4P5UED2TJTXLR6H2WXA62GE/graph.json","fetch_events":"https://pith.science/api/pith-number/NUS4P5UED2TJTXLR6H2WXA62GE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NUS4P5UED2TJTXLR6H2WXA62GE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NUS4P5UED2TJTXLR6H2WXA62GE/action/storage_attestation","attest_author":"https://pith.science/pith/NUS4P5UED2TJTXLR6H2WXA62GE/action/author_attestation","sign_citation":"https://pith.science/pith/NUS4P5UED2TJTXLR6H2WXA62GE/action/citation_signature","submit_replication":"https://pith.science/pith/NUS4P5UED2TJTXLR6H2WXA62GE/action/replication_record"}},"created_at":"2026-07-05T10:58:22.510419+00:00","updated_at":"2026-07-05T10:58:22.510419+00:00"}