{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:J6E5AMRIO4RNKBRTG7JP5WT3AW","short_pith_number":"pith:J6E5AMRI","schema_version":"1.0","canonical_sha256":"4f89d032287722d5063337d2feda7b05b5253865c8a8b904d3a409d1a82078c4","source":{"kind":"arxiv","id":"2507.12000","version":2},"attestation_state":"computed","paper":{"title":"DSSD: Efficient Edge-Device LLM Deployment and Collaborative Inference via Distributed Split Speculative Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.SP","authors_text":"Ce Zheng, Jiahong Ning, Tingting Yang","submitted_at":"2025-07-16T07:55:06Z","abstract_excerpt":"Large language models (LLMs) have transformed natural language processing but face critical deployment challenges in device-edge systems due to resource limitations and communication overhead. To address these issues, collaborative frameworks have emerged that combine small language models (SLMs) on devices with LLMs at the edge, using speculative decoding (SD) to improve efficiency. However, existing solutions often trade inference accuracy for latency or suffer from high uplink transmission costs when verifying candidate tokens. In this paper, we propose Distributed Split Speculative Decodin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.12000","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.SP","submitted_at":"2025-07-16T07:55:06Z","cross_cats_sorted":[],"title_canon_sha256":"a6e67d02c043ec51f88e0c5cd773890ec9b85216b387ae5f2d4e1f5968b830cf","abstract_canon_sha256":"7b6269bea4bcd8a2afc6f93dba52368e782cc00dd529229ca8475d48d706cda8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:38:32.241067Z","signature_b64":"PUTjlxGKZJFQSyyg9hK/CA83M+t44u/TwoDpVD6DcvNC/WGOhQ0cybyKJ5nf4qg8NfwePReQC2mxCN3uc6wYCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4f89d032287722d5063337d2feda7b05b5253865c8a8b904d3a409d1a82078c4","last_reissued_at":"2026-07-05T11:38:32.240503Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:38:32.240503Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DSSD: Efficient Edge-Device LLM Deployment and Collaborative Inference via Distributed Split Speculative Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.SP","authors_text":"Ce Zheng, Jiahong Ning, Tingting Yang","submitted_at":"2025-07-16T07:55:06Z","abstract_excerpt":"Large language models (LLMs) have transformed natural language processing but face critical deployment challenges in device-edge systems due to resource limitations and communication overhead. To address these issues, collaborative frameworks have emerged that combine small language models (SLMs) on devices with LLMs at the edge, using speculative decoding (SD) to improve efficiency. However, existing solutions often trade inference accuracy for latency or suffer from high uplink transmission costs when verifying candidate tokens. In this paper, we propose Distributed Split Speculative Decodin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.12000","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.12000/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.12000","created_at":"2026-07-05T11:38:32.240561+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.12000v2","created_at":"2026-07-05T11:38:32.240561+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.12000","created_at":"2026-07-05T11:38:32.240561+00:00"},{"alias_kind":"pith_short_12","alias_value":"J6E5AMRIO4RN","created_at":"2026-07-05T11:38:32.240561+00:00"},{"alias_kind":"pith_short_16","alias_value":"J6E5AMRIO4RNKBRT","created_at":"2026-07-05T11:38:32.240561+00:00"},{"alias_kind":"pith_short_8","alias_value":"J6E5AMRI","created_at":"2026-07-05T11:38:32.240561+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.11652","citing_title":"WISP: Waste- and Interference-Suppressed Distributed Speculative LLM Serving at the Edge via Dynamic Drafting and SLO-Aware Batching","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17701","citing_title":"WISV: Wireless-Informed Semantic Verification for Distributed Speculative Decoding in Device-Edge LLM Inference","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02218","citing_title":"CoVSpec: Efficient Device-Edge Co-Inference for Vision-Language Models via Speculative Decoding","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J6E5AMRIO4RNKBRTG7JP5WT3AW","json":"https://pith.science/pith/J6E5AMRIO4RNKBRTG7JP5WT3AW.json","graph_json":"https://pith.science/api/pith-number/J6E5AMRIO4RNKBRTG7JP5WT3AW/graph.json","events_json":"https://pith.science/api/pith-number/J6E5AMRIO4RNKBRTG7JP5WT3AW/events.json","paper":"https://pith.science/paper/J6E5AMRI"},"agent_actions":{"view_html":"https://pith.science/pith/J6E5AMRIO4RNKBRTG7JP5WT3AW","download_json":"https://pith.science/pith/J6E5AMRIO4RNKBRTG7JP5WT3AW.json","view_paper":"https://pith.science/paper/J6E5AMRI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.12000&json=true","fetch_graph":"https://pith.science/api/pith-number/J6E5AMRIO4RNKBRTG7JP5WT3AW/graph.json","fetch_events":"https://pith.science/api/pith-number/J6E5AMRIO4RNKBRTG7JP5WT3AW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J6E5AMRIO4RNKBRTG7JP5WT3AW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J6E5AMRIO4RNKBRTG7JP5WT3AW/action/storage_attestation","attest_author":"https://pith.science/pith/J6E5AMRIO4RNKBRTG7JP5WT3AW/action/author_attestation","sign_citation":"https://pith.science/pith/J6E5AMRIO4RNKBRTG7JP5WT3AW/action/citation_signature","submit_replication":"https://pith.science/pith/J6E5AMRIO4RNKBRTG7JP5WT3AW/action/replication_record"}},"created_at":"2026-07-05T11:38:32.240561+00:00","updated_at":"2026-07-05T11:38:32.240561+00:00"}