{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZZ3FKMSFJKE7A3QGMANDVLVUXY","short_pith_number":"pith:ZZ3FKMSF","schema_version":"1.0","canonical_sha256":"ce765532454a89f06e06601a3aaeb4be27c9c04a6a242006620889b1738fdae3","source":{"kind":"arxiv","id":"2502.11417","version":2},"attestation_state":"computed","paper":{"title":"DiSCo: Device-Server Collaborative LLM-Based Text Streaming Services","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Fan Lai, Penghan Wang, Ting Sun","submitted_at":"2025-02-17T04:15:45Z","abstract_excerpt":"The rapid rise of large language models (LLMs) in text streaming services has introduced significant cost and Quality of Experience (QoE) challenges in serving millions of daily requests, especially in meeting Time-To-First-Token (TTFT) and Time-Between-Token (TBT) requirements for real-time interactions. Our real-world measurements show that both server-based and on-device deployments struggle to meet diverse QoE demands: server deployments face high costs and last-hop issues (e.g., Internet latency and dynamics), while on-device LLM inference is constrained by resources.\n  We introduce DiSCo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.11417","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-17T04:15:45Z","cross_cats_sorted":["cs.DC"],"title_canon_sha256":"108402be26d55ab91fa149c2593e9cb59febd43a6aa292399dfcfb14f74a2232","abstract_canon_sha256":"099728ef2e2f4da7d096315ab82f0e682e35d8b5ad0aa7bd634b7c6560c31e76"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:08.466816Z","signature_b64":"fksyjuyOOdTuZpvFnT600QlYk9Jx7DJ7SsmtWc6nJzDjUAV7u4BnRO/CZso9P11ZMRngMPA+3Ce5+qFFl/f+CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce765532454a89f06e06601a3aaeb4be27c9c04a6a242006620889b1738fdae3","last_reissued_at":"2026-07-05T11:09:08.466183Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:08.466183Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DiSCo: Device-Server Collaborative LLM-Based Text Streaming Services","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Fan Lai, Penghan Wang, Ting Sun","submitted_at":"2025-02-17T04:15:45Z","abstract_excerpt":"The rapid rise of large language models (LLMs) in text streaming services has introduced significant cost and Quality of Experience (QoE) challenges in serving millions of daily requests, especially in meeting Time-To-First-Token (TTFT) and Time-Between-Token (TBT) requirements for real-time interactions. Our real-world measurements show that both server-based and on-device deployments struggle to meet diverse QoE demands: server deployments face high costs and last-hop issues (e.g., Internet latency and dynamics), while on-device LLM inference is constrained by resources.\n  We introduce DiSCo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.11417","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.11417/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.11417","created_at":"2026-07-05T11:09:08.466401+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.11417v2","created_at":"2026-07-05T11:09:08.466401+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.11417","created_at":"2026-07-05T11:09:08.466401+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZZ3FKMSFJKE7","created_at":"2026-07-05T11:09:08.466401+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZZ3FKMSFJKE7A3QG","created_at":"2026-07-05T11:09:08.466401+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZZ3FKMSF","created_at":"2026-07-05T11:09:08.466401+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.16731","citing_title":"Collaborative Inference and Learning between Edge SLMs and Cloud LLMs: A Survey of Algorithms, Execution, and Open Challenges","ref_index":41,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZZ3FKMSFJKE7A3QGMANDVLVUXY","json":"https://pith.science/pith/ZZ3FKMSFJKE7A3QGMANDVLVUXY.json","graph_json":"https://pith.science/api/pith-number/ZZ3FKMSFJKE7A3QGMANDVLVUXY/graph.json","events_json":"https://pith.science/api/pith-number/ZZ3FKMSFJKE7A3QGMANDVLVUXY/events.json","paper":"https://pith.science/paper/ZZ3FKMSF"},"agent_actions":{"view_html":"https://pith.science/pith/ZZ3FKMSFJKE7A3QGMANDVLVUXY","download_json":"https://pith.science/pith/ZZ3FKMSFJKE7A3QGMANDVLVUXY.json","view_paper":"https://pith.science/paper/ZZ3FKMSF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.11417&json=true","fetch_graph":"https://pith.science/api/pith-number/ZZ3FKMSFJKE7A3QGMANDVLVUXY/graph.json","fetch_events":"https://pith.science/api/pith-number/ZZ3FKMSFJKE7A3QGMANDVLVUXY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZZ3FKMSFJKE7A3QGMANDVLVUXY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZZ3FKMSFJKE7A3QGMANDVLVUXY/action/storage_attestation","attest_author":"https://pith.science/pith/ZZ3FKMSFJKE7A3QGMANDVLVUXY/action/author_attestation","sign_citation":"https://pith.science/pith/ZZ3FKMSFJKE7A3QGMANDVLVUXY/action/citation_signature","submit_replication":"https://pith.science/pith/ZZ3FKMSFJKE7A3QGMANDVLVUXY/action/replication_record"}},"created_at":"2026-07-05T11:09:08.466401+00:00","updated_at":"2026-07-05T11:09:08.466401+00:00"}