{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:U3IW7FHL7UK6SO4RPIOQGX6VLX","short_pith_number":"pith:U3IW7FHL","schema_version":"1.0","canonical_sha256":"a6d16f94ebfd15e93b917a1d035fd55df820d8578c91c77797f5cf559d49d1a8","source":{"kind":"arxiv","id":"2308.12066","version":3},"attestation_state":"computed","paper":{"title":"Pre-gated MoE: An Algorithm-System Co-Design for Fast and Scalable Mixture-of-Expert Inference","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.AR"],"primary_cat":"cs.LG","authors_text":"Changho Hwang, Jianyu Wei, Mao Yang, Ranggi Hwang, Shijie Cao, Ting Cao, Xiaohu Tang","submitted_at":"2023-08-23T11:25:37Z","abstract_excerpt":"Large language models (LLMs) based on transformers have made significant strides in recent years, the success of which is driven by scaling up their model size. Despite their high algorithmic performance, the computational and memory requirements of LLMs present unprecedented challenges. To tackle the high compute requirements of LLMs, the Mixture-of-Experts (MoE) architecture was introduced which is able to scale its model size without proportionally scaling up its computational requirements. Unfortunately, MoE's high memory demands and dynamic activation of sparse experts restrict its applic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.12066","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2023-08-23T11:25:37Z","cross_cats_sorted":["cs.AI","cs.AR"],"title_canon_sha256":"d4dede6b01d2826c58387ca640ca2da8632b8d631f9f5a8f82baae757f28f69b","abstract_canon_sha256":"6d6f6aded5ebb37ce4b5ac99131bb86246f8a724c80d9e3dd3aa4f486497fe43"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:12:44.571671Z","signature_b64":"aUUn269HpQ6v6HYAEFkLfCDIXFNEMIR0xatDKbHMAYonU8fvfZivBjA+LaQAErhZeNqOvlTIpR5OSMck65VeCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a6d16f94ebfd15e93b917a1d035fd55df820d8578c91c77797f5cf559d49d1a8","last_reissued_at":"2026-07-05T08:12:44.571233Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:12:44.571233Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pre-gated MoE: An Algorithm-System Co-Design for Fast and Scalable Mixture-of-Expert Inference","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.AR"],"primary_cat":"cs.LG","authors_text":"Changho Hwang, Jianyu Wei, Mao Yang, Ranggi Hwang, Shijie Cao, Ting Cao, Xiaohu Tang","submitted_at":"2023-08-23T11:25:37Z","abstract_excerpt":"Large language models (LLMs) based on transformers have made significant strides in recent years, the success of which is driven by scaling up their model size. Despite their high algorithmic performance, the computational and memory requirements of LLMs present unprecedented challenges. To tackle the high compute requirements of LLMs, the Mixture-of-Experts (MoE) architecture was introduced which is able to scale its model size without proportionally scaling up its computational requirements. Unfortunately, MoE's high memory demands and dynamic activation of sparse experts restrict its applic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.12066","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.12066/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.12066","created_at":"2026-07-05T08:12:44.571300+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.12066v3","created_at":"2026-07-05T08:12:44.571300+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.12066","created_at":"2026-07-05T08:12:44.571300+00:00"},{"alias_kind":"pith_short_12","alias_value":"U3IW7FHL7UK6","created_at":"2026-07-05T08:12:44.571300+00:00"},{"alias_kind":"pith_short_16","alias_value":"U3IW7FHL7UK6SO4R","created_at":"2026-07-05T08:12:44.571300+00:00"},{"alias_kind":"pith_short_8","alias_value":"U3IW7FHL","created_at":"2026-07-05T08:12:44.571300+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.22919","citing_title":"Hecto: Modular Sparse Experts for Adaptive and Interpretable Reasoning","ref_index":7,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U3IW7FHL7UK6SO4RPIOQGX6VLX","json":"https://pith.science/pith/U3IW7FHL7UK6SO4RPIOQGX6VLX.json","graph_json":"https://pith.science/api/pith-number/U3IW7FHL7UK6SO4RPIOQGX6VLX/graph.json","events_json":"https://pith.science/api/pith-number/U3IW7FHL7UK6SO4RPIOQGX6VLX/events.json","paper":"https://pith.science/paper/U3IW7FHL"},"agent_actions":{"view_html":"https://pith.science/pith/U3IW7FHL7UK6SO4RPIOQGX6VLX","download_json":"https://pith.science/pith/U3IW7FHL7UK6SO4RPIOQGX6VLX.json","view_paper":"https://pith.science/paper/U3IW7FHL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.12066&json=true","fetch_graph":"https://pith.science/api/pith-number/U3IW7FHL7UK6SO4RPIOQGX6VLX/graph.json","fetch_events":"https://pith.science/api/pith-number/U3IW7FHL7UK6SO4RPIOQGX6VLX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U3IW7FHL7UK6SO4RPIOQGX6VLX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U3IW7FHL7UK6SO4RPIOQGX6VLX/action/storage_attestation","attest_author":"https://pith.science/pith/U3IW7FHL7UK6SO4RPIOQGX6VLX/action/author_attestation","sign_citation":"https://pith.science/pith/U3IW7FHL7UK6SO4RPIOQGX6VLX/action/citation_signature","submit_replication":"https://pith.science/pith/U3IW7FHL7UK6SO4RPIOQGX6VLX/action/replication_record"}},"created_at":"2026-07-05T08:12:44.571300+00:00","updated_at":"2026-07-05T08:12:44.571300+00:00"}