{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5LW2AVIBKJSV7FZDL5OHTMYKB3","short_pith_number":"pith:5LW2AVIB","schema_version":"1.0","canonical_sha256":"eaeda0550152655f97235f5c79b30a0edd0f767d52239eebba5ce1c21684effa","source":{"kind":"arxiv","id":"2406.12295","version":2},"attestation_state":"computed","paper":{"title":"Fast and Slow Generating: An Empirical Study on Large and Small Language Models Collaborative Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Biqing Qi, Bowen Zhou, Ermo Hua, Jianyu Wang, Kaiyan Zhang, Ning Ding, Xingtai Lv","submitted_at":"2024-06-18T05:59:28Z","abstract_excerpt":"Large Language Models (LLMs) exhibit impressive capabilities across various applications but encounter substantial challenges such as high inference latency, considerable training costs, and the generation of hallucinations. Collaborative decoding between large and small language models (SLMs) presents a promising strategy to mitigate these issues through methods including speculative decoding, contrastive decoding, and emulator or proxy fine-tuning. However, the specifics of such collaborations, particularly from a unified perspective, remain largely unexplored. Inspired by dual-process cogni"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.12295","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-18T05:59:28Z","cross_cats_sorted":[],"title_canon_sha256":"a44ba2b9a289b2f088a5fb4b316764133e4d557a9a2d46aee843e4f8166f90ec","abstract_canon_sha256":"4295601e170fe2f201e7637290aabb1b8ba6262b709d9456ec6e19289b685bc6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:24:36.183268Z","signature_b64":"h5t9jVugM9Kh7uXnZfbTFEAJeb5vy1d4TS4qgH9Zz7lTk3Kk15oEn7K17vIoSfih9iS/rUUX7uv1CBelmPlcBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eaeda0550152655f97235f5c79b30a0edd0f767d52239eebba5ce1c21684effa","last_reissued_at":"2026-07-05T09:24:36.182763Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:24:36.182763Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fast and Slow Generating: An Empirical Study on Large and Small Language Models Collaborative Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Biqing Qi, Bowen Zhou, Ermo Hua, Jianyu Wang, Kaiyan Zhang, Ning Ding, Xingtai Lv","submitted_at":"2024-06-18T05:59:28Z","abstract_excerpt":"Large Language Models (LLMs) exhibit impressive capabilities across various applications but encounter substantial challenges such as high inference latency, considerable training costs, and the generation of hallucinations. Collaborative decoding between large and small language models (SLMs) presents a promising strategy to mitigate these issues through methods including speculative decoding, contrastive decoding, and emulator or proxy fine-tuning. However, the specifics of such collaborations, particularly from a unified perspective, remain largely unexplored. Inspired by dual-process cogni"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.12295","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.12295/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.12295","created_at":"2026-07-05T09:24:36.182823+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.12295v2","created_at":"2026-07-05T09:24:36.182823+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12295","created_at":"2026-07-05T09:24:36.182823+00:00"},{"alias_kind":"pith_short_12","alias_value":"5LW2AVIBKJSV","created_at":"2026-07-05T09:24:36.182823+00:00"},{"alias_kind":"pith_short_16","alias_value":"5LW2AVIBKJSV7FZD","created_at":"2026-07-05T09:24:36.182823+00:00"},{"alias_kind":"pith_short_8","alias_value":"5LW2AVIB","created_at":"2026-07-05T09:24:36.182823+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.16731","citing_title":"Collaborative Inference and Learning between Edge SLMs and Cloud LLMs: A Survey of Algorithms, Execution, and Open Challenges","ref_index":84,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5LW2AVIBKJSV7FZDL5OHTMYKB3","json":"https://pith.science/pith/5LW2AVIBKJSV7FZDL5OHTMYKB3.json","graph_json":"https://pith.science/api/pith-number/5LW2AVIBKJSV7FZDL5OHTMYKB3/graph.json","events_json":"https://pith.science/api/pith-number/5LW2AVIBKJSV7FZDL5OHTMYKB3/events.json","paper":"https://pith.science/paper/5LW2AVIB"},"agent_actions":{"view_html":"https://pith.science/pith/5LW2AVIBKJSV7FZDL5OHTMYKB3","download_json":"https://pith.science/pith/5LW2AVIBKJSV7FZDL5OHTMYKB3.json","view_paper":"https://pith.science/paper/5LW2AVIB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.12295&json=true","fetch_graph":"https://pith.science/api/pith-number/5LW2AVIBKJSV7FZDL5OHTMYKB3/graph.json","fetch_events":"https://pith.science/api/pith-number/5LW2AVIBKJSV7FZDL5OHTMYKB3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5LW2AVIBKJSV7FZDL5OHTMYKB3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5LW2AVIBKJSV7FZDL5OHTMYKB3/action/storage_attestation","attest_author":"https://pith.science/pith/5LW2AVIBKJSV7FZDL5OHTMYKB3/action/author_attestation","sign_citation":"https://pith.science/pith/5LW2AVIBKJSV7FZDL5OHTMYKB3/action/citation_signature","submit_replication":"https://pith.science/pith/5LW2AVIBKJSV7FZDL5OHTMYKB3/action/replication_record"}},"created_at":"2026-07-05T09:24:36.182823+00:00","updated_at":"2026-07-05T09:24:36.182823+00:00"}