{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:J5C5PD475R2RT27QMPQPMOQUTI","short_pith_number":"pith:J5C5PD47","schema_version":"1.0","canonical_sha256":"4f45d78f9fec7519ebf063e0f63a149a3772791b5f29642c60e223d2174b1934","source":{"kind":"arxiv","id":"2502.01662","version":2},"attestation_state":"computed","paper":{"title":"Fast Large Language Model Collaborative Decoding via Speculation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jiale Fu, Jiaming Fan, Junkai Chen, Xin Geng, Xu Yang, Yuchu Jiang","submitted_at":"2025-02-01T05:22:11Z","abstract_excerpt":"Large Language Model (LLM) collaborative decoding techniques improve output quality by combining the outputs of multiple models at each generation step, but they incur high computational costs. In this paper, we introduce Collaborative decoding via Speculation (CoS), a novel framework that accelerates collaborative decoding without compromising performance. Inspired by Speculative Decoding--where a small proposal model generates tokens sequentially, and a larger target model verifies them in parallel, our approach builds on two key insights: (1) the verification distribution can be the combine"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01662","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-01T05:22:11Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"5e55c3065b0993acb97620935ce2056e9eade62d4a15cd46920cc6d0cd328e1b","abstract_canon_sha256":"7620ec0fd7e1f813e2f59f7c75be0e2e61fb37c4e9acd522c429e68cdca1715a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:40.428326Z","signature_b64":"q0wREDi315NdotwXWxMzspJXkA/aQrnWem1sr9LiDRl5QaVuZfPyOr7u5r7yODnTWhRNDLBh8DuycFGmYm16CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4f45d78f9fec7519ebf063e0f63a149a3772791b5f29642c60e223d2174b1934","last_reissued_at":"2026-07-05T11:11:40.427766Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:40.427766Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fast Large Language Model Collaborative Decoding via Speculation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jiale Fu, Jiaming Fan, Junkai Chen, Xin Geng, Xu Yang, Yuchu Jiang","submitted_at":"2025-02-01T05:22:11Z","abstract_excerpt":"Large Language Model (LLM) collaborative decoding techniques improve output quality by combining the outputs of multiple models at each generation step, but they incur high computational costs. In this paper, we introduce Collaborative decoding via Speculation (CoS), a novel framework that accelerates collaborative decoding without compromising performance. Inspired by Speculative Decoding--where a small proposal model generates tokens sequentially, and a larger target model verifies them in parallel, our approach builds on two key insights: (1) the verification distribution can be the combine"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01662","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01662/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01662","created_at":"2026-07-05T11:11:40.427817+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01662v2","created_at":"2026-07-05T11:11:40.427817+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01662","created_at":"2026-07-05T11:11:40.427817+00:00"},{"alias_kind":"pith_short_12","alias_value":"J5C5PD475R2R","created_at":"2026-07-05T11:11:40.427817+00:00"},{"alias_kind":"pith_short_16","alias_value":"J5C5PD475R2RT27Q","created_at":"2026-07-05T11:11:40.427817+00:00"},{"alias_kind":"pith_short_8","alias_value":"J5C5PD47","created_at":"2026-07-05T11:11:40.427817+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.00419","citing_title":"Rethinking LLM Ensembling from the Perspective of Mixture Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25777","citing_title":"SpecFed: Accelerating Federated LLM Inference with Speculative Decoding and Compressed Transmission","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18562","citing_title":"AnchorSeg: Language Grounded Query Banks for Reasoning Segmentation","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J5C5PD475R2RT27QMPQPMOQUTI","json":"https://pith.science/pith/J5C5PD475R2RT27QMPQPMOQUTI.json","graph_json":"https://pith.science/api/pith-number/J5C5PD475R2RT27QMPQPMOQUTI/graph.json","events_json":"https://pith.science/api/pith-number/J5C5PD475R2RT27QMPQPMOQUTI/events.json","paper":"https://pith.science/paper/J5C5PD47"},"agent_actions":{"view_html":"https://pith.science/pith/J5C5PD475R2RT27QMPQPMOQUTI","download_json":"https://pith.science/pith/J5C5PD475R2RT27QMPQPMOQUTI.json","view_paper":"https://pith.science/paper/J5C5PD47","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01662&json=true","fetch_graph":"https://pith.science/api/pith-number/J5C5PD475R2RT27QMPQPMOQUTI/graph.json","fetch_events":"https://pith.science/api/pith-number/J5C5PD475R2RT27QMPQPMOQUTI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J5C5PD475R2RT27QMPQPMOQUTI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J5C5PD475R2RT27QMPQPMOQUTI/action/storage_attestation","attest_author":"https://pith.science/pith/J5C5PD475R2RT27QMPQPMOQUTI/action/author_attestation","sign_citation":"https://pith.science/pith/J5C5PD475R2RT27QMPQPMOQUTI/action/citation_signature","submit_replication":"https://pith.science/pith/J5C5PD475R2RT27QMPQPMOQUTI/action/replication_record"}},"created_at":"2026-07-05T11:11:40.427817+00:00","updated_at":"2026-07-05T11:11:40.427817+00:00"}