{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LJZ7BEFOJQUX73XE62TS6Z4D4B","short_pith_number":"pith:LJZ7BEFO","schema_version":"1.0","canonical_sha256":"5a73f090ae4c297feee4f6a72f6783e05f44510ee4709769ee1af482605d8237","source":{"kind":"arxiv","id":"2507.21276","version":1},"attestation_state":"computed","paper":{"title":"LeMix: Unified Scheduling for LLM Training and Inference on Multi-GPU Systems","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL","cs.DC"],"primary_cat":"cs.AI","authors_text":"Cong Liu, Yinglun Zhu, Yufei Li, Zexin Li","submitted_at":"2025-07-28T19:03:26Z","abstract_excerpt":"Modern deployment of large language models (LLMs) frequently involves both inference serving and continuous retraining to stay aligned with evolving data and user feedback. Common practices separate these workloads onto distinct servers in isolated phases, causing substantial inefficiencies (e.g., GPU idleness) and delayed adaptation to new data in distributed settings. Our empirical analysis reveals that these inefficiencies stem from dynamic request arrivals during serving and workload heterogeneity in pipeline-parallel training. To address these challenges, we propose LeMix, a system for co"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.21276","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2025-07-28T19:03:26Z","cross_cats_sorted":["cs.CL","cs.DC"],"title_canon_sha256":"d42d7242594493a725a9f24b547a002cdd1acba7ed837424ef8767bd97da6bc8","abstract_canon_sha256":"d81f36ef911797c97415f56773e998f6819ac88a2ec3322edc8e56baac2b6bdd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:44:56.163312Z","signature_b64":"L0iM1j52nCSGHIy3HSMAAeiL12XuA8Bm6xl9p/m3hXx2/H8g1abe6nOM6wAXhiOR8djPkgHMcczmi5XOCDStCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5a73f090ae4c297feee4f6a72f6783e05f44510ee4709769ee1af482605d8237","last_reissued_at":"2026-07-05T11:44:56.162860Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:44:56.162860Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LeMix: Unified Scheduling for LLM Training and Inference on Multi-GPU Systems","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL","cs.DC"],"primary_cat":"cs.AI","authors_text":"Cong Liu, Yinglun Zhu, Yufei Li, Zexin Li","submitted_at":"2025-07-28T19:03:26Z","abstract_excerpt":"Modern deployment of large language models (LLMs) frequently involves both inference serving and continuous retraining to stay aligned with evolving data and user feedback. Common practices separate these workloads onto distinct servers in isolated phases, causing substantial inefficiencies (e.g., GPU idleness) and delayed adaptation to new data in distributed settings. Our empirical analysis reveals that these inefficiencies stem from dynamic request arrivals during serving and workload heterogeneity in pipeline-parallel training. To address these challenges, we propose LeMix, a system for co"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.21276","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.21276/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.21276","created_at":"2026-07-05T11:44:56.162915+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.21276v1","created_at":"2026-07-05T11:44:56.162915+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.21276","created_at":"2026-07-05T11:44:56.162915+00:00"},{"alias_kind":"pith_short_12","alias_value":"LJZ7BEFOJQUX","created_at":"2026-07-05T11:44:56.162915+00:00"},{"alias_kind":"pith_short_16","alias_value":"LJZ7BEFOJQUX73XE","created_at":"2026-07-05T11:44:56.162915+00:00"},{"alias_kind":"pith_short_8","alias_value":"LJZ7BEFO","created_at":"2026-07-05T11:44:56.162915+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24044","citing_title":"RED: Adaptive Real-Time DAG Scheduling for Robotic Inference under Environmental Dynamics","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23027","citing_title":"PIMbot: A Self-Adaptive Attack Framework for Adversarial Manipulation of Multi-Robot Reinforcement Learning","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LJZ7BEFOJQUX73XE62TS6Z4D4B","json":"https://pith.science/pith/LJZ7BEFOJQUX73XE62TS6Z4D4B.json","graph_json":"https://pith.science/api/pith-number/LJZ7BEFOJQUX73XE62TS6Z4D4B/graph.json","events_json":"https://pith.science/api/pith-number/LJZ7BEFOJQUX73XE62TS6Z4D4B/events.json","paper":"https://pith.science/paper/LJZ7BEFO"},"agent_actions":{"view_html":"https://pith.science/pith/LJZ7BEFOJQUX73XE62TS6Z4D4B","download_json":"https://pith.science/pith/LJZ7BEFOJQUX73XE62TS6Z4D4B.json","view_paper":"https://pith.science/paper/LJZ7BEFO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.21276&json=true","fetch_graph":"https://pith.science/api/pith-number/LJZ7BEFOJQUX73XE62TS6Z4D4B/graph.json","fetch_events":"https://pith.science/api/pith-number/LJZ7BEFOJQUX73XE62TS6Z4D4B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LJZ7BEFOJQUX73XE62TS6Z4D4B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LJZ7BEFOJQUX73XE62TS6Z4D4B/action/storage_attestation","attest_author":"https://pith.science/pith/LJZ7BEFOJQUX73XE62TS6Z4D4B/action/author_attestation","sign_citation":"https://pith.science/pith/LJZ7BEFOJQUX73XE62TS6Z4D4B/action/citation_signature","submit_replication":"https://pith.science/pith/LJZ7BEFOJQUX73XE62TS6Z4D4B/action/replication_record"}},"created_at":"2026-07-05T11:44:56.162915+00:00","updated_at":"2026-07-05T11:44:56.162915+00:00"}