{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XG4YKYKFGCHS5GP7X2DMWDTATS","short_pith_number":"pith:XG4YKYKF","schema_version":"1.0","canonical_sha256":"b9b9856145308f2e99ffbe86cb0e609ca7938deb5884396e0beb2c25dcc47654","source":{"kind":"arxiv","id":"2407.00029","version":1},"attestation_state":"computed","paper":{"title":"Distributed Inference Performance Optimization for LLMs on CPUs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Changqing Li, Chen Meng, Duyi Wang, Pujiang He, Shan Zhou, Sheng Gui, Weifei Yu, Wenhuan Huang","submitted_at":"2024-05-16T08:39:37Z","abstract_excerpt":"Large language models (LLMs) hold tremendous potential for addressing numerous real-world challenges, yet they typically demand significant computational resources and memory. Deploying LLMs onto a resource-limited hardware device with restricted memory capacity presents considerable challenges. Distributed computing emerges as a prevalent strategy to mitigate single-node memory constraints and expedite LLM inference performance. To reduce the hardware limitation burden, we proposed an efficient distributed inference optimization solution for LLMs on CPUs. We conduct experiments with the propo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.00029","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2024-05-16T08:39:37Z","cross_cats_sorted":[],"title_canon_sha256":"ccbbbee280a7b4443b9d923e7d8ec0b3a3e006bd0a3a1b66e86fc20c74958274","abstract_canon_sha256":"9094bf97652a67275e81fec0d20ba805fce3f64cc6a4e5b63288bf371607094e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:38:13.388013Z","signature_b64":"/uVlxpOK+NP1V9R2fBN25krxS5B+3qTN3/rlAT6aqdnHFTuvLTw17EBIt+GR3Y0ZA7QKNpBkLyqayEw76nfZDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9b9856145308f2e99ffbe86cb0e609ca7938deb5884396e0beb2c25dcc47654","last_reissued_at":"2026-07-05T08:38:13.387576Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:38:13.387576Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Distributed Inference Performance Optimization for LLMs on CPUs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Changqing Li, Chen Meng, Duyi Wang, Pujiang He, Shan Zhou, Sheng Gui, Weifei Yu, Wenhuan Huang","submitted_at":"2024-05-16T08:39:37Z","abstract_excerpt":"Large language models (LLMs) hold tremendous potential for addressing numerous real-world challenges, yet they typically demand significant computational resources and memory. Deploying LLMs onto a resource-limited hardware device with restricted memory capacity presents considerable challenges. Distributed computing emerges as a prevalent strategy to mitigate single-node memory constraints and expedite LLM inference performance. To reduce the hardware limitation burden, we proposed an efficient distributed inference optimization solution for LLMs on CPUs. We conduct experiments with the propo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.00029","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.00029/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.00029","created_at":"2026-07-05T08:38:13.387631+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.00029v1","created_at":"2026-07-05T08:38:13.387631+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.00029","created_at":"2026-07-05T08:38:13.387631+00:00"},{"alias_kind":"pith_short_12","alias_value":"XG4YKYKFGCHS","created_at":"2026-07-05T08:38:13.387631+00:00"},{"alias_kind":"pith_short_16","alias_value":"XG4YKYKFGCHS5GP7","created_at":"2026-07-05T08:38:13.387631+00:00"},{"alias_kind":"pith_short_8","alias_value":"XG4YKYKF","created_at":"2026-07-05T08:38:13.387631+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.05313","citing_title":"Optimizing Distributed Deployment of Mixture-of-Experts Model Inference in Serverless Computing","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XG4YKYKFGCHS5GP7X2DMWDTATS","json":"https://pith.science/pith/XG4YKYKFGCHS5GP7X2DMWDTATS.json","graph_json":"https://pith.science/api/pith-number/XG4YKYKFGCHS5GP7X2DMWDTATS/graph.json","events_json":"https://pith.science/api/pith-number/XG4YKYKFGCHS5GP7X2DMWDTATS/events.json","paper":"https://pith.science/paper/XG4YKYKF"},"agent_actions":{"view_html":"https://pith.science/pith/XG4YKYKFGCHS5GP7X2DMWDTATS","download_json":"https://pith.science/pith/XG4YKYKFGCHS5GP7X2DMWDTATS.json","view_paper":"https://pith.science/paper/XG4YKYKF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.00029&json=true","fetch_graph":"https://pith.science/api/pith-number/XG4YKYKFGCHS5GP7X2DMWDTATS/graph.json","fetch_events":"https://pith.science/api/pith-number/XG4YKYKFGCHS5GP7X2DMWDTATS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XG4YKYKFGCHS5GP7X2DMWDTATS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XG4YKYKFGCHS5GP7X2DMWDTATS/action/storage_attestation","attest_author":"https://pith.science/pith/XG4YKYKFGCHS5GP7X2DMWDTATS/action/author_attestation","sign_citation":"https://pith.science/pith/XG4YKYKFGCHS5GP7X2DMWDTATS/action/citation_signature","submit_replication":"https://pith.science/pith/XG4YKYKFGCHS5GP7X2DMWDTATS/action/replication_record"}},"created_at":"2026-07-05T08:38:13.387631+00:00","updated_at":"2026-07-05T08:38:13.387631+00:00"}