{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LQ67J327Y7FCBDZMKQVT7BBEEL","short_pith_number":"pith:LQ67J327","schema_version":"1.0","canonical_sha256":"5c3df4ef5fc7ca208f2c542b3f842422f951e6ef191224a942afb6d32c5175a2","source":{"kind":"arxiv","id":"2401.11181","version":1},"attestation_state":"computed","paper":{"title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Chenxi Wang, Cunchen Hu, Hao Feng, Heyang Huang, Jiang Xu, Liangliang Xu, Ninghui Sun, Sa Wang, Shuang Chen, Xusheng Chen, Yizhou Shan, Yungang Bao","submitted_at":"2024-01-20T09:43:36Z","abstract_excerpt":"Transformer-based large language model (LLM) inference serving is now the backbone of many cloud services. LLM inference consists of a prefill phase and a decode phase. However, existing LLM deployment practices often overlook the distinct characteristics of these phases, leading to significant interference. To mitigate interference, our insight is to carefully schedule and group inference requests based on their characteristics. We realize this idea in TetriInfer through three pillars. First, it partitions prompts into fixed-size chunks so that the accelerator always runs close to its computa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.11181","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2024-01-20T09:43:36Z","cross_cats_sorted":[],"title_canon_sha256":"d3a72040bd1c172a4fa9938cb59d7622d79a8c6bce1e43194803f6b3f547bf47","abstract_canon_sha256":"ed8802b9c11510eed3006ebc8ab3f699a9e9a5e7c8e6a70baff78f38ef0e1172"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:35:46.980169Z","signature_b64":"sIzWcUMqxpD9D9djknbwTYE6zs5d6eOn0RKUwdqsGFGSsaFwpQOw4fc6R/j7ybN8itSntSeEkOdm//ljDzBJCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5c3df4ef5fc7ca208f2c542b3f842422f951e6ef191224a942afb6d32c5175a2","last_reissued_at":"2026-07-05T07:35:46.979718Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:35:46.979718Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Inference without Interference: Disaggregate LLM Inference for Mixed Downstream Workloads","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Chenxi Wang, Cunchen Hu, Hao Feng, Heyang Huang, Jiang Xu, Liangliang Xu, Ninghui Sun, Sa Wang, Shuang Chen, Xusheng Chen, Yizhou Shan, Yungang Bao","submitted_at":"2024-01-20T09:43:36Z","abstract_excerpt":"Transformer-based large language model (LLM) inference serving is now the backbone of many cloud services. LLM inference consists of a prefill phase and a decode phase. However, existing LLM deployment practices often overlook the distinct characteristics of these phases, leading to significant interference. To mitigate interference, our insight is to carefully schedule and group inference requests based on their characteristics. We realize this idea in TetriInfer through three pillars. First, it partitions prompts into fixed-size chunks so that the accelerator always runs close to its computa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.11181","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.11181/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.11181","created_at":"2026-07-05T07:35:46.979773+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.11181v1","created_at":"2026-07-05T07:35:46.979773+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.11181","created_at":"2026-07-05T07:35:46.979773+00:00"},{"alias_kind":"pith_short_12","alias_value":"LQ67J327Y7FC","created_at":"2026-07-05T07:35:46.979773+00:00"},{"alias_kind":"pith_short_16","alias_value":"LQ67J327Y7FCBDZM","created_at":"2026-07-05T07:35:46.979773+00:00"},{"alias_kind":"pith_short_8","alias_value":"LQ67J327","created_at":"2026-07-05T07:35:46.979773+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22327","citing_title":"Geometry-Aware Online Scheduling for LLM Serving: From Theoretical Bound to System Practice","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02043","citing_title":"Towards Load-Aware Prefill Deflection for Disaggregated LLM Serving","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29708","citing_title":"Demystifying the Design Space and Best Practices for Heterogeneous LLM Inference and Serving","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20577","citing_title":"Human-Less LLM Serving: Quantifying the Human Tax on Throughput","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29708","citing_title":"Demystifying the Design Space and Best Practices for Heterogeneous LLM Inference and Serving","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23389","citing_title":"AlignedServe: Orchestrating Prefix-aware Batching to Build a High-throughput and Computing-efficient LLM Serving System","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17613","citing_title":"VeriCache: Turning Lossy KV Cache into Lossless LLM Inference","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2510.11938","citing_title":"FlexPipe: Adapting Dynamic LLM Serving Through Inflight Pipeline Refactoring in Fragmented Serverless Clusters","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13668","citing_title":"STAR: Decode-Phase Rescheduling for LLM Inference","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2512.09427","citing_title":"ODMA: On-Demand Memory Allocation Strategy for LLM Serving on LPDDR-Class Accelerators","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":273,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04357","citing_title":"Coral: Cost-Efficient Multi-LLM Serving over Heterogeneous Cloud GPUs","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LQ67J327Y7FCBDZMKQVT7BBEEL","json":"https://pith.science/pith/LQ67J327Y7FCBDZMKQVT7BBEEL.json","graph_json":"https://pith.science/api/pith-number/LQ67J327Y7FCBDZMKQVT7BBEEL/graph.json","events_json":"https://pith.science/api/pith-number/LQ67J327Y7FCBDZMKQVT7BBEEL/events.json","paper":"https://pith.science/paper/LQ67J327"},"agent_actions":{"view_html":"https://pith.science/pith/LQ67J327Y7FCBDZMKQVT7BBEEL","download_json":"https://pith.science/pith/LQ67J327Y7FCBDZMKQVT7BBEEL.json","view_paper":"https://pith.science/paper/LQ67J327","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.11181&json=true","fetch_graph":"https://pith.science/api/pith-number/LQ67J327Y7FCBDZMKQVT7BBEEL/graph.json","fetch_events":"https://pith.science/api/pith-number/LQ67J327Y7FCBDZMKQVT7BBEEL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LQ67J327Y7FCBDZMKQVT7BBEEL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LQ67J327Y7FCBDZMKQVT7BBEEL/action/storage_attestation","attest_author":"https://pith.science/pith/LQ67J327Y7FCBDZMKQVT7BBEEL/action/author_attestation","sign_citation":"https://pith.science/pith/LQ67J327Y7FCBDZMKQVT7BBEEL/action/citation_signature","submit_replication":"https://pith.science/pith/LQ67J327Y7FCBDZMKQVT7BBEEL/action/replication_record"}},"created_at":"2026-07-05T07:35:46.979773+00:00","updated_at":"2026-07-05T07:35:46.979773+00:00"}