{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5BCOFNCMOWFTONFBVS5PZPW5EM","short_pith_number":"pith:5BCOFNCM","schema_version":"1.0","canonical_sha256":"e844e2b44c758b3734a1acbafcbedd2302e796aeab5a8efd794401617d8a2ae2","source":{"kind":"arxiv","id":"2505.01572","version":1},"attestation_state":"computed","paper":{"title":"PipeSpec: Breaking Stage Dependencies in Hierarchical LLM Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.AI","authors_text":"Bradley McDanel, Sai Qian Zhang, Yunhai Hu, Zining Liu","submitted_at":"2025-05-02T20:29:31Z","abstract_excerpt":"Speculative decoding accelerates large language model inference by using smaller draft models to generate candidate tokens for parallel verification. However, current approaches are limited by sequential stage dependencies that prevent full hardware utilization. We present PipeSpec, a framework that generalizes speculative decoding to $k$ models arranged in a hierarchical pipeline, enabling asynchronous execution with lightweight coordination for prediction verification and rollback. Our analytical model characterizes token generation rates across pipeline stages and proves guaranteed throughp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.01572","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-02T20:29:31Z","cross_cats_sorted":["cs.DC"],"title_canon_sha256":"b37970ab1e672e7fbe4b7d8be7bb95b413ff62e54156a6c4fc6977c5e07b48f0","abstract_canon_sha256":"6a6eb610344c59569db58e33bf15acb04bf2363cae5fc22ccdfc4c222cffe06d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:58:00.121412Z","signature_b64":"3Zh2ALFK2SB1zhHeJKEP0OoR3tV8HItrRSaELLj422eWvp2ftNjP85ditZRxYZbcumDEGsPeE0WrGbg9MGamAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e844e2b44c758b3734a1acbafcbedd2302e796aeab5a8efd794401617d8a2ae2","last_reissued_at":"2026-07-05T10:58:00.120984Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:58:00.120984Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PipeSpec: Breaking Stage Dependencies in Hierarchical LLM Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.AI","authors_text":"Bradley McDanel, Sai Qian Zhang, Yunhai Hu, Zining Liu","submitted_at":"2025-05-02T20:29:31Z","abstract_excerpt":"Speculative decoding accelerates large language model inference by using smaller draft models to generate candidate tokens for parallel verification. However, current approaches are limited by sequential stage dependencies that prevent full hardware utilization. We present PipeSpec, a framework that generalizes speculative decoding to $k$ models arranged in a hierarchical pipeline, enabling asynchronous execution with lightweight coordination for prediction verification and rollback. Our analytical model characterizes token generation rates across pipeline stages and proves guaranteed throughp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.01572","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.01572/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.01572","created_at":"2026-07-05T10:58:00.121039+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.01572v1","created_at":"2026-07-05T10:58:00.121039+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.01572","created_at":"2026-07-05T10:58:00.121039+00:00"},{"alias_kind":"pith_short_12","alias_value":"5BCOFNCMOWFT","created_at":"2026-07-05T10:58:00.121039+00:00"},{"alias_kind":"pith_short_16","alias_value":"5BCOFNCMOWFTONFB","created_at":"2026-07-05T10:58:00.121039+00:00"},{"alias_kind":"pith_short_8","alias_value":"5BCOFNCM","created_at":"2026-07-05T10:58:00.121039+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.16731","citing_title":"Collaborative Inference and Learning between Edge SLMs and Cloud LLMs: A Survey of Algorithms, Execution, and Open Challenges","ref_index":135,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5BCOFNCMOWFTONFBVS5PZPW5EM","json":"https://pith.science/pith/5BCOFNCMOWFTONFBVS5PZPW5EM.json","graph_json":"https://pith.science/api/pith-number/5BCOFNCMOWFTONFBVS5PZPW5EM/graph.json","events_json":"https://pith.science/api/pith-number/5BCOFNCMOWFTONFBVS5PZPW5EM/events.json","paper":"https://pith.science/paper/5BCOFNCM"},"agent_actions":{"view_html":"https://pith.science/pith/5BCOFNCMOWFTONFBVS5PZPW5EM","download_json":"https://pith.science/pith/5BCOFNCMOWFTONFBVS5PZPW5EM.json","view_paper":"https://pith.science/paper/5BCOFNCM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.01572&json=true","fetch_graph":"https://pith.science/api/pith-number/5BCOFNCMOWFTONFBVS5PZPW5EM/graph.json","fetch_events":"https://pith.science/api/pith-number/5BCOFNCMOWFTONFBVS5PZPW5EM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5BCOFNCMOWFTONFBVS5PZPW5EM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5BCOFNCMOWFTONFBVS5PZPW5EM/action/storage_attestation","attest_author":"https://pith.science/pith/5BCOFNCMOWFTONFBVS5PZPW5EM/action/author_attestation","sign_citation":"https://pith.science/pith/5BCOFNCMOWFTONFBVS5PZPW5EM/action/citation_signature","submit_replication":"https://pith.science/pith/5BCOFNCMOWFTONFBVS5PZPW5EM/action/replication_record"}},"created_at":"2026-07-05T10:58:00.121039+00:00","updated_at":"2026-07-05T10:58:00.121039+00:00"}