{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IZYT6PUOLLXOEKM7OHOPD3IVKK","short_pith_number":"pith:IZYT6PUO","schema_version":"1.0","canonical_sha256":"46713f3e8e5aeee2299f71dcf1ed1552be402628e5d913b5d2fee6ee7b3996c4","source":{"kind":"arxiv","id":"2412.18934","version":2},"attestation_state":"computed","paper":{"title":"Dovetail: A CPU/GPU Heterogeneous Speculative Decoding for LLM inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baizhou Xu, Dongsheng Li, Libo Zhang, Rui Li, Songzhu Mei, Zhaoning Zhang, Zhiliang Tian","submitted_at":"2024-12-25T15:45:18Z","abstract_excerpt":"With the continuous advancement in the performance of large language models (LLMs), their demand for computational resources and memory has significantly increased, which poses major challenges for efficient inference on consumer-grade devices and legacy servers. These devices typically feature relatively weaker GPUs and stronger CPUs. Although techniques such as parameter offloading and partial offloading can alleviate GPU memory pressure to some extent, their effectiveness is limited due to communication latency and suboptimal hardware resource utilization. To address this issue, we propose "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.18934","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-12-25T15:45:18Z","cross_cats_sorted":[],"title_canon_sha256":"234722d7fb92d3e8ffc3ee527dce33d2d2da5150805ee11c4cea174ddb76f840","abstract_canon_sha256":"766b8548546d163539afcfcc6f4e576bce97e31055658bf1e5b1a5162b815f01"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:07:02.661425Z","signature_b64":"bGp9BvOxywHZMejMqI6X6nVCyKEbr1Ph++Gtv0qUdimEUhJDRr7SrW2/PEoYPBHwIB/n/2lvrSG1VzwFaiodBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"46713f3e8e5aeee2299f71dcf1ed1552be402628e5d913b5d2fee6ee7b3996c4","last_reissued_at":"2026-07-05T12:07:02.660993Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:07:02.660993Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dovetail: A CPU/GPU Heterogeneous Speculative Decoding for LLM inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baizhou Xu, Dongsheng Li, Libo Zhang, Rui Li, Songzhu Mei, Zhaoning Zhang, Zhiliang Tian","submitted_at":"2024-12-25T15:45:18Z","abstract_excerpt":"With the continuous advancement in the performance of large language models (LLMs), their demand for computational resources and memory has significantly increased, which poses major challenges for efficient inference on consumer-grade devices and legacy servers. These devices typically feature relatively weaker GPUs and stronger CPUs. Although techniques such as parameter offloading and partial offloading can alleviate GPU memory pressure to some extent, their effectiveness is limited due to communication latency and suboptimal hardware resource utilization. To address this issue, we propose "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.18934","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.18934/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.18934","created_at":"2026-07-05T12:07:02.661045+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.18934v2","created_at":"2026-07-05T12:07:02.661045+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.18934","created_at":"2026-07-05T12:07:02.661045+00:00"},{"alias_kind":"pith_short_12","alias_value":"IZYT6PUOLLXO","created_at":"2026-07-05T12:07:02.661045+00:00"},{"alias_kind":"pith_short_16","alias_value":"IZYT6PUOLLXOEKM7","created_at":"2026-07-05T12:07:02.661045+00:00"},{"alias_kind":"pith_short_8","alias_value":"IZYT6PUO","created_at":"2026-07-05T12:07:02.661045+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12303","citing_title":"From 2D Grids to 1D Tokens: Reforming Shared Representations for Multimodal Image Fusion","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IZYT6PUOLLXOEKM7OHOPD3IVKK","json":"https://pith.science/pith/IZYT6PUOLLXOEKM7OHOPD3IVKK.json","graph_json":"https://pith.science/api/pith-number/IZYT6PUOLLXOEKM7OHOPD3IVKK/graph.json","events_json":"https://pith.science/api/pith-number/IZYT6PUOLLXOEKM7OHOPD3IVKK/events.json","paper":"https://pith.science/paper/IZYT6PUO"},"agent_actions":{"view_html":"https://pith.science/pith/IZYT6PUOLLXOEKM7OHOPD3IVKK","download_json":"https://pith.science/pith/IZYT6PUOLLXOEKM7OHOPD3IVKK.json","view_paper":"https://pith.science/paper/IZYT6PUO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.18934&json=true","fetch_graph":"https://pith.science/api/pith-number/IZYT6PUOLLXOEKM7OHOPD3IVKK/graph.json","fetch_events":"https://pith.science/api/pith-number/IZYT6PUOLLXOEKM7OHOPD3IVKK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IZYT6PUOLLXOEKM7OHOPD3IVKK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IZYT6PUOLLXOEKM7OHOPD3IVKK/action/storage_attestation","attest_author":"https://pith.science/pith/IZYT6PUOLLXOEKM7OHOPD3IVKK/action/author_attestation","sign_citation":"https://pith.science/pith/IZYT6PUOLLXOEKM7OHOPD3IVKK/action/citation_signature","submit_replication":"https://pith.science/pith/IZYT6PUOLLXOEKM7OHOPD3IVKK/action/replication_record"}},"created_at":"2026-07-05T12:07:02.661045+00:00","updated_at":"2026-07-05T12:07:02.661045+00:00"}