{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BKCSA6KYFABL2ZLIFUO4IHNRRT","short_pith_number":"pith:BKCSA6KY","schema_version":"1.0","canonical_sha256":"0a852079582802bd65682d1dc41db18ccdddba4530ef955e0b8816f4909941f7","source":{"kind":"arxiv","id":"2404.16283","version":2},"attestation_state":"computed","paper":{"title":"Andes: Defining and Enhancing Quality-of-Experience in LLM-Based Text Streaming Services","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Fan Lai, Jae-Won Chung, Jiachen Liu, Mosharaf Chowdhury, Myungjin Lee, Zhiyu Wu","submitted_at":"2024-04-25T01:56:00Z","abstract_excerpt":"Large language models (LLMs) are now at the core of conversational AI services such as real-time translation and chatbots, which provide live user interaction by incrementally streaming text to the user. However, existing LLM serving systems fail to provide good user experience because their optimization metrics are not always aligned with user experience.\n  In this paper, we first introduce and define the notion of Quality-of-Experience (QoE) for text streaming services by considering each user's end-to-end interaction timeline. Based on this, we propose Andes, a QoE-aware LLM serving system "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.16283","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.DC","submitted_at":"2024-04-25T01:56:00Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"ecd78f9cf4f3696cd4a9478524577a754370b6ae46ba21181089c2ec6a636514","abstract_canon_sha256":"45e4b2bec491654c64b5eadaf67d582b88b8ec9a6ab151da561a9c54ec40a22d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:48:32.540987Z","signature_b64":"MO0ete9B52Xh3oefs6CJI/UTseHIGhCl0UL/W1ERV1F8O3Ggs1Clyx7uHae14HPiYHNkQPobP3G3DwBmixM5BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0a852079582802bd65682d1dc41db18ccdddba4530ef955e0b8816f4909941f7","last_reissued_at":"2026-07-05T09:48:32.540452Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:48:32.540452Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Andes: Defining and Enhancing Quality-of-Experience in LLM-Based Text Streaming Services","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Fan Lai, Jae-Won Chung, Jiachen Liu, Mosharaf Chowdhury, Myungjin Lee, Zhiyu Wu","submitted_at":"2024-04-25T01:56:00Z","abstract_excerpt":"Large language models (LLMs) are now at the core of conversational AI services such as real-time translation and chatbots, which provide live user interaction by incrementally streaming text to the user. However, existing LLM serving systems fail to provide good user experience because their optimization metrics are not always aligned with user experience.\n  In this paper, we first introduce and define the notion of Quality-of-Experience (QoE) for text streaming services by considering each user's end-to-end interaction timeline. Based on this, we propose Andes, a QoE-aware LLM serving system "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.16283","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.16283/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.16283","created_at":"2026-07-05T09:48:32.540516+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.16283v2","created_at":"2026-07-05T09:48:32.540516+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.16283","created_at":"2026-07-05T09:48:32.540516+00:00"},{"alias_kind":"pith_short_12","alias_value":"BKCSA6KYFABL","created_at":"2026-07-05T09:48:32.540516+00:00"},{"alias_kind":"pith_short_16","alias_value":"BKCSA6KYFABL2ZLI","created_at":"2026-07-05T09:48:32.540516+00:00"},{"alias_kind":"pith_short_8","alias_value":"BKCSA6KY","created_at":"2026-07-05T09:48:32.540516+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22983","citing_title":"LiveServe: Interaction-Aware Serving for Real-Time Omni-Modal LLMs","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23001","citing_title":"EnerInfer: Energy-Aware On-Device LLM Inference","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18431","citing_title":"Beyond Prediction: Tail-Aware Scheduling for LLM Inference","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2504.16397","citing_title":"Compass: SLO-aware Query Planner for Compound AI Serving at Scale","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2505.09999","citing_title":"ServeGen: Workload Characterization and Generation of Large Language Model Serving in Production","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25080","citing_title":"CacheFlow: Efficient LLM Serving with 3D-Parallel KV Cache Restoration","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06914","citing_title":"Regulating Branch Parallelism in LLM Serving","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BKCSA6KYFABL2ZLIFUO4IHNRRT","json":"https://pith.science/pith/BKCSA6KYFABL2ZLIFUO4IHNRRT.json","graph_json":"https://pith.science/api/pith-number/BKCSA6KYFABL2ZLIFUO4IHNRRT/graph.json","events_json":"https://pith.science/api/pith-number/BKCSA6KYFABL2ZLIFUO4IHNRRT/events.json","paper":"https://pith.science/paper/BKCSA6KY"},"agent_actions":{"view_html":"https://pith.science/pith/BKCSA6KYFABL2ZLIFUO4IHNRRT","download_json":"https://pith.science/pith/BKCSA6KYFABL2ZLIFUO4IHNRRT.json","view_paper":"https://pith.science/paper/BKCSA6KY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.16283&json=true","fetch_graph":"https://pith.science/api/pith-number/BKCSA6KYFABL2ZLIFUO4IHNRRT/graph.json","fetch_events":"https://pith.science/api/pith-number/BKCSA6KYFABL2ZLIFUO4IHNRRT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BKCSA6KYFABL2ZLIFUO4IHNRRT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BKCSA6KYFABL2ZLIFUO4IHNRRT/action/storage_attestation","attest_author":"https://pith.science/pith/BKCSA6KYFABL2ZLIFUO4IHNRRT/action/author_attestation","sign_citation":"https://pith.science/pith/BKCSA6KYFABL2ZLIFUO4IHNRRT/action/citation_signature","submit_replication":"https://pith.science/pith/BKCSA6KYFABL2ZLIFUO4IHNRRT/action/replication_record"}},"created_at":"2026-07-05T09:48:32.540516+00:00","updated_at":"2026-07-05T09:48:32.540516+00:00"}