{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4CIBK464NEDYGENMAIQ3XNACW6","short_pith_number":"pith:4CIBK464","schema_version":"1.0","canonical_sha256":"e0901573dc69078311ac0221bbb402b79a2251b30ab0cc7e7d43268eb350577b","source":{"kind":"arxiv","id":"2408.05499","version":1},"attestation_state":"computed","paper":{"title":"LLMServingSim: A HW/SW Co-Simulation Infrastructure for LLM Inference Serving at Scale","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Guseul Heo, Hyunmin Choi, Jaehong Cho, Jongse Park, Minsu Kim","submitted_at":"2024-08-10T09:26:15Z","abstract_excerpt":"Recently, there has been an extensive research effort in building efficient large language model (LLM) inference serving systems. These efforts not only include innovations in the algorithm and software domains but also constitute developments of various hardware acceleration techniques. Nevertheless, there is a lack of simulation infrastructure capable of accurately modeling versatile hardware-software behaviors in LLM serving systems without extensively extending the simulation time. This paper aims to develop an effective simulation tool, called LLMServingSim, to support future research in "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.05499","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2024-08-10T09:26:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1aa92481914a7b5698b5aebc1ceffdb9269c8bc50e77e469c53e1561d6575bc0","abstract_canon_sha256":"602bbcc5cbb003dcb5a1ab31c4e20b75050d0a2764f953df5e0c71e15723a533"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:40.316088Z","signature_b64":"oBDMB7IFyqbbj6z9Q3Hipf0oe+cg5WLH6CWCd8/42ocvdLGilkCaYFcVgC4N0//TRdAiRYbFLustQFIeMV5QBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e0901573dc69078311ac0221bbb402b79a2251b30ab0cc7e7d43268eb350577b","last_reissued_at":"2026-07-05T09:41:40.315584Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:40.315584Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLMServingSim: A HW/SW Co-Simulation Infrastructure for LLM Inference Serving at Scale","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Guseul Heo, Hyunmin Choi, Jaehong Cho, Jongse Park, Minsu Kim","submitted_at":"2024-08-10T09:26:15Z","abstract_excerpt":"Recently, there has been an extensive research effort in building efficient large language model (LLM) inference serving systems. These efforts not only include innovations in the algorithm and software domains but also constitute developments of various hardware acceleration techniques. Nevertheless, there is a lack of simulation infrastructure capable of accurately modeling versatile hardware-software behaviors in LLM serving systems without extensively extending the simulation time. This paper aims to develop an effective simulation tool, called LLMServingSim, to support future research in "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.05499","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.05499/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.05499","created_at":"2026-07-05T09:41:40.315643+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.05499v1","created_at":"2026-07-05T09:41:40.315643+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.05499","created_at":"2026-07-05T09:41:40.315643+00:00"},{"alias_kind":"pith_short_12","alias_value":"4CIBK464NEDY","created_at":"2026-07-05T09:41:40.315643+00:00"},{"alias_kind":"pith_short_16","alias_value":"4CIBK464NEDYGENM","created_at":"2026-07-05T09:41:40.315643+00:00"},{"alias_kind":"pith_short_8","alias_value":"4CIBK464","created_at":"2026-07-05T09:41:40.315643+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2504.09775","citing_title":"MIST: A Co-Design Framework for Heterogeneous, Multi-Stage LLM Inference","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12090","citing_title":"Evaluating Cross-Architecture Performance Modeling of Distributed ML Workloads Using StableHLO","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4CIBK464NEDYGENMAIQ3XNACW6","json":"https://pith.science/pith/4CIBK464NEDYGENMAIQ3XNACW6.json","graph_json":"https://pith.science/api/pith-number/4CIBK464NEDYGENMAIQ3XNACW6/graph.json","events_json":"https://pith.science/api/pith-number/4CIBK464NEDYGENMAIQ3XNACW6/events.json","paper":"https://pith.science/paper/4CIBK464"},"agent_actions":{"view_html":"https://pith.science/pith/4CIBK464NEDYGENMAIQ3XNACW6","download_json":"https://pith.science/pith/4CIBK464NEDYGENMAIQ3XNACW6.json","view_paper":"https://pith.science/paper/4CIBK464","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.05499&json=true","fetch_graph":"https://pith.science/api/pith-number/4CIBK464NEDYGENMAIQ3XNACW6/graph.json","fetch_events":"https://pith.science/api/pith-number/4CIBK464NEDYGENMAIQ3XNACW6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4CIBK464NEDYGENMAIQ3XNACW6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4CIBK464NEDYGENMAIQ3XNACW6/action/storage_attestation","attest_author":"https://pith.science/pith/4CIBK464NEDYGENMAIQ3XNACW6/action/author_attestation","sign_citation":"https://pith.science/pith/4CIBK464NEDYGENMAIQ3XNACW6/action/citation_signature","submit_replication":"https://pith.science/pith/4CIBK464NEDYGENMAIQ3XNACW6/action/replication_record"}},"created_at":"2026-07-05T09:41:40.315643+00:00","updated_at":"2026-07-05T09:41:40.315643+00:00"}