{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:US3QXFH4GBAD3BFJJCAVWFSZD7","short_pith_number":"pith:US3QXFH4","schema_version":"1.0","canonical_sha256":"a4b70b94fc30403d84a948815b16591ff109f5053c919c67a087ae9b0076cc97","source":{"kind":"arxiv","id":"2607.27735","version":1},"attestation_state":"computed","paper":{"title":"A Sparse Glimpse of the Whole: Train-Free Self-Speculative Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Min Lyu, Ruilin Liu, Yinlong Xu, Yuan Zeng, Yuesong Liu, Yu Guo","submitted_at":"2026-07-30T06:16:56Z","abstract_excerpt":"Speculative decoding alleviates the memory-bandwidth bottleneck in large language model inference, but its acceleration is jointly constrained by drafting overhead, token acceptance, and speculation length. We present a unified efficiency analysis showing that extending the speculation horizon can reduce rather than improve speedup when the marginal acceptance probability falls below the relative drafting cost. Guided by this analysis, we introduce SparseSpec-L, a training-free self-speculative decoding framework for long-context inference. SparseSpec-L generates lightweight drafts directly fr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.27735","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-07-30T06:16:56Z","cross_cats_sorted":[],"title_canon_sha256":"56efc75f99b77471810c10b7f44b8c8682341bfa7d4d9755ae8a585fd0899e33","abstract_canon_sha256":"02301f9b0327e1d1c34b354299958d447fde36f18d8e84745e9144be11f3dbde"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a4b70b94fc30403d84a948815b16591ff109f5053c919c67a087ae9b0076cc97","last_reissued_at":"2026-07-31T01:30:05.956579Z","signature_status":"unsigned_v0","first_computed_at":"2026-07-31T01:30:05.956579Z"},"graph_snapshot":{"paper":{"title":"A Sparse Glimpse of the Whole: Train-Free Self-Speculative Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Min Lyu, Ruilin Liu, Yinlong Xu, Yuan Zeng, Yuesong Liu, Yu Guo","submitted_at":"2026-07-30T06:16:56Z","abstract_excerpt":"Speculative decoding alleviates the memory-bandwidth bottleneck in large language model inference, but its acceleration is jointly constrained by drafting overhead, token acceptance, and speculation length. We present a unified efficiency analysis showing that extending the speculation horizon can reduce rather than improve speedup when the marginal acceptance probability falls below the relative drafting cost. Guided by this analysis, we introduce SparseSpec-L, a training-free self-speculative decoding framework for long-context inference. SparseSpec-L generates lightweight drafts directly fr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.27735","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.27735/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.27735","created_at":"2026-07-31T01:30:05.959938+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.27735v1","created_at":"2026-07-31T01:30:05.959938+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.27735","created_at":"2026-07-31T01:30:05.959938+00:00"},{"alias_kind":"pith_short_12","alias_value":"US3QXFH4GBAD","created_at":"2026-07-31T01:30:05.959938+00:00"},{"alias_kind":"pith_short_16","alias_value":"US3QXFH4GBAD3BFJ","created_at":"2026-07-31T01:30:05.959938+00:00"},{"alias_kind":"pith_short_8","alias_value":"US3QXFH4","created_at":"2026-07-31T01:30:05.959938+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/US3QXFH4GBAD3BFJJCAVWFSZD7","json":"https://pith.science/pith/US3QXFH4GBAD3BFJJCAVWFSZD7.json","graph_json":"https://pith.science/api/pith-number/US3QXFH4GBAD3BFJJCAVWFSZD7/graph.json","events_json":"https://pith.science/api/pith-number/US3QXFH4GBAD3BFJJCAVWFSZD7/events.json","paper":"https://pith.science/paper/US3QXFH4"},"agent_actions":{"view_html":"https://pith.science/pith/US3QXFH4GBAD3BFJJCAVWFSZD7","download_json":"https://pith.science/pith/US3QXFH4GBAD3BFJJCAVWFSZD7.json","view_paper":"https://pith.science/paper/US3QXFH4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.27735&json=true","fetch_graph":"https://pith.science/api/pith-number/US3QXFH4GBAD3BFJJCAVWFSZD7/graph.json","fetch_events":"https://pith.science/api/pith-number/US3QXFH4GBAD3BFJJCAVWFSZD7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/US3QXFH4GBAD3BFJJCAVWFSZD7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/US3QXFH4GBAD3BFJJCAVWFSZD7/action/storage_attestation","attest_author":"https://pith.science/pith/US3QXFH4GBAD3BFJJCAVWFSZD7/action/author_attestation","sign_citation":"https://pith.science/pith/US3QXFH4GBAD3BFJJCAVWFSZD7/action/citation_signature","submit_replication":"https://pith.science/pith/US3QXFH4GBAD3BFJJCAVWFSZD7/action/replication_record"}},"created_at":"2026-07-31T01:30:05.959938+00:00","updated_at":"2026-07-31T01:30:05.959938+00:00"}