{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UR7ZX7AY33J5GM3WZROUI6JJVW","short_pith_number":"pith:UR7ZX7AY","schema_version":"1.0","canonical_sha256":"a47f9bfc18ded3d33376cc5d447929adaf09e9039996c403f975da65c15e1995","source":{"kind":"arxiv","id":"2402.05109","version":2},"attestation_state":"computed","paper":{"title":"Hydra: Sequentially-Dependent Draft Heads for Medusa Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aniruddha Nrusimha, Christopher Rinard, Jonathan Ragan-Kelley, Rishab Parthasarathy, William Brandon, Zachary Ankner","submitted_at":"2024-02-07T18:58:50Z","abstract_excerpt":"To combat the memory bandwidth-bound nature of autoregressive LLM inference, previous research has proposed the speculative decoding frame-work. To perform speculative decoding, a small draft model proposes candidate continuations of the input sequence that are then verified in parallel by the base model. One way to specify the draft model, as used in the recent Medusa decoding framework, is as a collection of lightweight heads, called draft heads, that operate on the base model's hidden states. To date, all existing draft heads have been sequentially independent, meaning that they speculate t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.05109","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-07T18:58:50Z","cross_cats_sorted":[],"title_canon_sha256":"450ad928dfe6b963939367348a02e550583a0fb5eb1779ab1d5f2c0a105d125d","abstract_canon_sha256":"8f6fccf4f87feee3175d2c638a3df4cc59ae4b2f1d3718589f471c43af592599"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:16:46.362358Z","signature_b64":"n3P2KCh+ESl2j993vrFJgS2y6S5CozHc/BanlUwoRbMEISnE6FCAkqcjbv6bu795M1Tt8CrUiMwng3xMYsA8BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a47f9bfc18ded3d33376cc5d447929adaf09e9039996c403f975da65c15e1995","last_reissued_at":"2026-07-05T09:16:46.361869Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:16:46.361869Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hydra: Sequentially-Dependent Draft Heads for Medusa Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aniruddha Nrusimha, Christopher Rinard, Jonathan Ragan-Kelley, Rishab Parthasarathy, William Brandon, Zachary Ankner","submitted_at":"2024-02-07T18:58:50Z","abstract_excerpt":"To combat the memory bandwidth-bound nature of autoregressive LLM inference, previous research has proposed the speculative decoding frame-work. To perform speculative decoding, a small draft model proposes candidate continuations of the input sequence that are then verified in parallel by the base model. One way to specify the draft model, as used in the recent Medusa decoding framework, is as a collection of lightweight heads, called draft heads, that operate on the base model's hidden states. To date, all existing draft heads have been sequentially independent, meaning that they speculate t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.05109","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.05109/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.05109","created_at":"2026-07-05T09:16:46.361927+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.05109v2","created_at":"2026-07-05T09:16:46.361927+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.05109","created_at":"2026-07-05T09:16:46.361927+00:00"},{"alias_kind":"pith_short_12","alias_value":"UR7ZX7AY33J5","created_at":"2026-07-05T09:16:46.361927+00:00"},{"alias_kind":"pith_short_16","alias_value":"UR7ZX7AY33J5GM3W","created_at":"2026-07-05T09:16:46.361927+00:00"},{"alias_kind":"pith_short_8","alias_value":"UR7ZX7AY","created_at":"2026-07-05T09:16:46.361927+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07409","citing_title":"DeLS-Spec: Decoupled Long-Short Contexts for Parallel Speculative Drafting","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24447","citing_title":"P-MTP: Efficient Document Parsing via Multi-Token Prediction with Progressive Depth Scaling","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26744","citing_title":"HyperDFlash: Hyper-Connection-Aligned Block Speculative Decoding with Gated Residual Reduction","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17518","citing_title":"SpecGen: Accelerating Agentic Kernel Optimization with Speculative Generation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27550","citing_title":"EntMTP: Accelerating LLM Inference with Entropy Guided Multi Token Prediction","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26744","citing_title":"HyperDFlash: Hyper-Connection-Aligned Block Speculative Decoding with Gated Residual Reduction","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29707","citing_title":"Domino: Decoupling Causal Modeling from Autoregressive Drafting in Speculative Decoding","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00535","citing_title":"DREAM-S: Speculative Decoding with Searchable Drafting and Target-Aware Refinement for Multimodal Generation","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07243","citing_title":"SpecBlock: Block-Iterative Speculative Decoding with Dynamic Tree Drafting","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18810","citing_title":"D-PACE: Dynamic Position-Aware Cross-Entropy for Parallel Speculative Drafting","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14005","citing_title":"Mistletoe: Stealthy Acceleration-Collapse Attacks on Speculative Decoding","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20022","citing_title":"FlexDraft: Flexible Speculative Decoding via Attention Tuning and Bonus-Guided Calibration","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14978","citing_title":"Performance-Driven Policy Optimization for Speculative Decoding with Adaptive Windowing","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14005","citing_title":"Mistletoe: Stealthy Acceleration-Collapse Attacks on Speculative Decoding","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26469","citing_title":"An Empirical Study of Speculative Decoding on Software Engineering Tasks","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08632","citing_title":"PARD-2: Target-Aligned Parallel Draft Model for Dual-Mode Speculative Decoding","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08862","citing_title":"BubbleSpec: Turning Long-Tail Bubbles into Speculative Rollout Drafts for Synchronous Reinforcement Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10453","citing_title":"SlimSpec: Low-Rank Draft LM-Head for Accelerated Speculative Decoding","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07243","citing_title":"SpecBlock: Block-Iterative Speculative Decoding with Dynamic Tree Drafting","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UR7ZX7AY33J5GM3WZROUI6JJVW","json":"https://pith.science/pith/UR7ZX7AY33J5GM3WZROUI6JJVW.json","graph_json":"https://pith.science/api/pith-number/UR7ZX7AY33J5GM3WZROUI6JJVW/graph.json","events_json":"https://pith.science/api/pith-number/UR7ZX7AY33J5GM3WZROUI6JJVW/events.json","paper":"https://pith.science/paper/UR7ZX7AY"},"agent_actions":{"view_html":"https://pith.science/pith/UR7ZX7AY33J5GM3WZROUI6JJVW","download_json":"https://pith.science/pith/UR7ZX7AY33J5GM3WZROUI6JJVW.json","view_paper":"https://pith.science/paper/UR7ZX7AY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.05109&json=true","fetch_graph":"https://pith.science/api/pith-number/UR7ZX7AY33J5GM3WZROUI6JJVW/graph.json","fetch_events":"https://pith.science/api/pith-number/UR7ZX7AY33J5GM3WZROUI6JJVW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UR7ZX7AY33J5GM3WZROUI6JJVW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UR7ZX7AY33J5GM3WZROUI6JJVW/action/storage_attestation","attest_author":"https://pith.science/pith/UR7ZX7AY33J5GM3WZROUI6JJVW/action/author_attestation","sign_citation":"https://pith.science/pith/UR7ZX7AY33J5GM3WZROUI6JJVW/action/citation_signature","submit_replication":"https://pith.science/pith/UR7ZX7AY33J5GM3WZROUI6JJVW/action/replication_record"}},"created_at":"2026-07-05T09:16:46.361927+00:00","updated_at":"2026-07-05T09:16:46.361927+00:00"}