{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AORZQYTONMXEBG24RLRYUDYXHJ","short_pith_number":"pith:AORZQYTO","schema_version":"1.0","canonical_sha256":"03a398626e6b2e409b5c8ae38a0f173a4079c29c60d4c4ab9a91fc9375fa9d62","source":{"kind":"arxiv","id":"2507.18007","version":1},"attestation_state":"computed","paper":{"title":"Cloud Native System for LLM Inference Serving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Chengzhong Xu, Jingfeng Wu, Junhan Liao, Kejiang Ye, Minxian Xu, Yiyuan He","submitted_at":"2025-07-24T00:49:56Z","abstract_excerpt":"Large Language Models (LLMs) are revolutionizing numerous industries, but their substantial computational demands create challenges for efficient deployment, particularly in cloud environments. Traditional approaches to inference serving often struggle with resource inefficiencies, leading to high operational costs, latency issues, and limited scalability. This article explores how Cloud Native technologies, such as containerization, microservices, and dynamic scheduling, can fundamentally improve LLM inference serving. By leveraging these technologies, we demonstrate how a Cloud Native system"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.18007","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2025-07-24T00:49:56Z","cross_cats_sorted":[],"title_canon_sha256":"468645c2355b21f8be9566dc846c77319a130a37d0a34e9f5f0a1e7b547c6464","abstract_canon_sha256":"1bb5f4eff918fbecc98534ab117b3234a854ab9cb371036748cd41750130e17f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:42:38.992133Z","signature_b64":"lk34SDNfXm7tVOCINlwvgQhdk4aPx3neix85a54k+XwPXglqanj53FDts0v9wQqI4pOu+OnOMSw2xpRXYudKAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"03a398626e6b2e409b5c8ae38a0f173a4079c29c60d4c4ab9a91fc9375fa9d62","last_reissued_at":"2026-07-05T11:42:38.991623Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:42:38.991623Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cloud Native System for LLM Inference Serving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Chengzhong Xu, Jingfeng Wu, Junhan Liao, Kejiang Ye, Minxian Xu, Yiyuan He","submitted_at":"2025-07-24T00:49:56Z","abstract_excerpt":"Large Language Models (LLMs) are revolutionizing numerous industries, but their substantial computational demands create challenges for efficient deployment, particularly in cloud environments. Traditional approaches to inference serving often struggle with resource inefficiencies, leading to high operational costs, latency issues, and limited scalability. This article explores how Cloud Native technologies, such as containerization, microservices, and dynamic scheduling, can fundamentally improve LLM inference serving. By leveraging these technologies, we demonstrate how a Cloud Native system"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.18007","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.18007/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.18007","created_at":"2026-07-05T11:42:38.991685+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.18007v1","created_at":"2026-07-05T11:42:38.991685+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.18007","created_at":"2026-07-05T11:42:38.991685+00:00"},{"alias_kind":"pith_short_12","alias_value":"AORZQYTONMXE","created_at":"2026-07-05T11:42:38.991685+00:00"},{"alias_kind":"pith_short_16","alias_value":"AORZQYTONMXEBG24","created_at":"2026-07-05T11:42:38.991685+00:00"},{"alias_kind":"pith_short_8","alias_value":"AORZQYTO","created_at":"2026-07-05T11:42:38.991685+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.17227","citing_title":"Cloud-native and Distributed Systems for Efficient and Scalable Large Language Models -- A Research Agenda","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AORZQYTONMXEBG24RLRYUDYXHJ","json":"https://pith.science/pith/AORZQYTONMXEBG24RLRYUDYXHJ.json","graph_json":"https://pith.science/api/pith-number/AORZQYTONMXEBG24RLRYUDYXHJ/graph.json","events_json":"https://pith.science/api/pith-number/AORZQYTONMXEBG24RLRYUDYXHJ/events.json","paper":"https://pith.science/paper/AORZQYTO"},"agent_actions":{"view_html":"https://pith.science/pith/AORZQYTONMXEBG24RLRYUDYXHJ","download_json":"https://pith.science/pith/AORZQYTONMXEBG24RLRYUDYXHJ.json","view_paper":"https://pith.science/paper/AORZQYTO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.18007&json=true","fetch_graph":"https://pith.science/api/pith-number/AORZQYTONMXEBG24RLRYUDYXHJ/graph.json","fetch_events":"https://pith.science/api/pith-number/AORZQYTONMXEBG24RLRYUDYXHJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AORZQYTONMXEBG24RLRYUDYXHJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AORZQYTONMXEBG24RLRYUDYXHJ/action/storage_attestation","attest_author":"https://pith.science/pith/AORZQYTONMXEBG24RLRYUDYXHJ/action/author_attestation","sign_citation":"https://pith.science/pith/AORZQYTONMXEBG24RLRYUDYXHJ/action/citation_signature","submit_replication":"https://pith.science/pith/AORZQYTONMXEBG24RLRYUDYXHJ/action/replication_record"}},"created_at":"2026-07-05T11:42:38.991685+00:00","updated_at":"2026-07-05T11:42:38.991685+00:00"}