{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:NHJFATVXR35HFU5ZG3NUE6GN5A","short_pith_number":"pith:NHJFATVX","schema_version":"1.0","canonical_sha256":"69d2504eb78efa72d3b936db4278cde82e1d8290dd3bfae32cd6c2eee1def7f4","source":{"kind":"arxiv","id":"2607.14647","version":1},"attestation_state":"computed","paper":{"title":"D-cut: Adaptive Verification Depth Pruning for Batched Speculative Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Guanghua Yu, Guangshuo Qin, Hong Liu, Jianchen Zhu, Jiebin Zhang, Junhan Shi, Rui Cen, Song Liu, Tianyu Liu, Yuhao Shen","submitted_at":"2026-07-16T07:18:05Z","abstract_excerpt":"Speculative decoding accelerates large language model (LLM) inference without compromising output quality. Recent parallel drafting methods further improve single-request performance by decoupling draft length from drafting latency, enabling longer drafts and higher mean accepted tokens (MAT). However, under high request concurrency, long drafts waste substantial computation on rejected tokens, increasing verification cost and potentially making speculative decoding slower than autoregressive decoding. We present D-Cut, an adaptive pruning method that selects draft tokens jointly across the ba"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.14647","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-07-16T07:18:05Z","cross_cats_sorted":[],"title_canon_sha256":"528f147de8b6b3212f9ede41dcfd2a62c7b17693c2d4b4723c33aa75d42118ec","abstract_canon_sha256":"94b94b911f84f8595c1d4faafdb870cd44d09044650616b1a6fa5eb9308adbf2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-17T01:21:22.833872Z","signature_b64":"Mv/LUPS5DoiK6883je7Y70t92KXzQqPGiKsTGOzWOnsIWzso73Bxr0B/+OpJio903+CMXACXflEb/ZpfumHgDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"69d2504eb78efa72d3b936db4278cde82e1d8290dd3bfae32cd6c2eee1def7f4","last_reissued_at":"2026-07-17T01:21:22.833056Z","signature_status":"signed_v1","first_computed_at":"2026-07-17T01:21:22.833056Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"D-cut: Adaptive Verification Depth Pruning for Batched Speculative Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Guanghua Yu, Guangshuo Qin, Hong Liu, Jianchen Zhu, Jiebin Zhang, Junhan Shi, Rui Cen, Song Liu, Tianyu Liu, Yuhao Shen","submitted_at":"2026-07-16T07:18:05Z","abstract_excerpt":"Speculative decoding accelerates large language model (LLM) inference without compromising output quality. Recent parallel drafting methods further improve single-request performance by decoupling draft length from drafting latency, enabling longer drafts and higher mean accepted tokens (MAT). However, under high request concurrency, long drafts waste substantial computation on rejected tokens, increasing verification cost and potentially making speculative decoding slower than autoregressive decoding. We present D-Cut, an adaptive pruning method that selects draft tokens jointly across the ba"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.14647","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.14647/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.14647","created_at":"2026-07-17T01:21:22.833473+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.14647v1","created_at":"2026-07-17T01:21:22.833473+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.14647","created_at":"2026-07-17T01:21:22.833473+00:00"},{"alias_kind":"pith_short_12","alias_value":"NHJFATVXR35H","created_at":"2026-07-17T01:21:22.833473+00:00"},{"alias_kind":"pith_short_16","alias_value":"NHJFATVXR35HFU5Z","created_at":"2026-07-17T01:21:22.833473+00:00"},{"alias_kind":"pith_short_8","alias_value":"NHJFATVX","created_at":"2026-07-17T01:21:22.833473+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NHJFATVXR35HFU5ZG3NUE6GN5A","json":"https://pith.science/pith/NHJFATVXR35HFU5ZG3NUE6GN5A.json","graph_json":"https://pith.science/api/pith-number/NHJFATVXR35HFU5ZG3NUE6GN5A/graph.json","events_json":"https://pith.science/api/pith-number/NHJFATVXR35HFU5ZG3NUE6GN5A/events.json","paper":"https://pith.science/paper/NHJFATVX"},"agent_actions":{"view_html":"https://pith.science/pith/NHJFATVXR35HFU5ZG3NUE6GN5A","download_json":"https://pith.science/pith/NHJFATVXR35HFU5ZG3NUE6GN5A.json","view_paper":"https://pith.science/paper/NHJFATVX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.14647&json=true","fetch_graph":"https://pith.science/api/pith-number/NHJFATVXR35HFU5ZG3NUE6GN5A/graph.json","fetch_events":"https://pith.science/api/pith-number/NHJFATVXR35HFU5ZG3NUE6GN5A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NHJFATVXR35HFU5ZG3NUE6GN5A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NHJFATVXR35HFU5ZG3NUE6GN5A/action/storage_attestation","attest_author":"https://pith.science/pith/NHJFATVXR35HFU5ZG3NUE6GN5A/action/author_attestation","sign_citation":"https://pith.science/pith/NHJFATVXR35HFU5ZG3NUE6GN5A/action/citation_signature","submit_replication":"https://pith.science/pith/NHJFATVXR35HFU5ZG3NUE6GN5A/action/replication_record"}},"created_at":"2026-07-17T01:21:22.833473+00:00","updated_at":"2026-07-17T01:21:22.833473+00:00"}