{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:J5677BAQOTZ6HQ3423WKSFWMF3","short_pith_number":"pith:J5677BAQ","schema_version":"1.0","canonical_sha256":"4f7dff841074f3e3c37cd6eca916cc2ec154697412bc3ec291bffbf783d3433b","source":{"kind":"arxiv","id":"2504.07878","version":1},"attestation_state":"computed","paper":{"title":"Token Level Routing Inference System for Edge Devices","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.CL","authors_text":"Eric Xing, Hongyi Wang, Huaxiu Yao, Jianshu She, Qirong Ho, Wenhao Zheng, Zhengzhong Liu","submitted_at":"2025-04-10T15:54:19Z","abstract_excerpt":"The computational complexity of large language model (LLM) inference significantly constrains their deployment efficiency on edge devices. In contrast, small language models offer faster decoding and lower resource consumption but often suffer from degraded response quality and heightened susceptibility to hallucinations. To address this trade-off, collaborative decoding, in which a large model assists in generating critical tokens, has emerged as a promising solution. This paradigm leverages the strengths of both model types by enabling high-quality inference through selective intervention of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.07878","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-10T15:54:19Z","cross_cats_sorted":["cs.DC"],"title_canon_sha256":"6125376dadc567e2d0024130dc14e355353c00b548dd9776651405970bd0955f","abstract_canon_sha256":"9b3f5c91d3d86d25daf4fde9a033fd22c626b20a99d89b457baae8d84f60a428"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:21.349499Z","signature_b64":"74STn7JAoJpBHkg443HPiaLEYyceHB+Doy/JOL1ZIQSkfXVGtgmtw6FpvbLqc0ogzo9e89bzPOTNGWcrV4CcDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4f7dff841074f3e3c37cd6eca916cc2ec154697412bc3ec291bffbf783d3433b","last_reissued_at":"2026-07-05T10:47:21.349032Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:21.349032Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Token Level Routing Inference System for Edge Devices","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.CL","authors_text":"Eric Xing, Hongyi Wang, Huaxiu Yao, Jianshu She, Qirong Ho, Wenhao Zheng, Zhengzhong Liu","submitted_at":"2025-04-10T15:54:19Z","abstract_excerpt":"The computational complexity of large language model (LLM) inference significantly constrains their deployment efficiency on edge devices. In contrast, small language models offer faster decoding and lower resource consumption but often suffer from degraded response quality and heightened susceptibility to hallucinations. To address this trade-off, collaborative decoding, in which a large model assists in generating critical tokens, has emerged as a promising solution. This paradigm leverages the strengths of both model types by enabling high-quality inference through selective intervention of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.07878","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.07878/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.07878","created_at":"2026-07-05T10:47:21.349085+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.07878v1","created_at":"2026-07-05T10:47:21.349085+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.07878","created_at":"2026-07-05T10:47:21.349085+00:00"},{"alias_kind":"pith_short_12","alias_value":"J5677BAQOTZ6","created_at":"2026-07-05T10:47:21.349085+00:00"},{"alias_kind":"pith_short_16","alias_value":"J5677BAQOTZ6HQ34","created_at":"2026-07-05T10:47:21.349085+00:00"},{"alias_kind":"pith_short_8","alias_value":"J5677BAQ","created_at":"2026-07-05T10:47:21.349085+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J5677BAQOTZ6HQ3423WKSFWMF3","json":"https://pith.science/pith/J5677BAQOTZ6HQ3423WKSFWMF3.json","graph_json":"https://pith.science/api/pith-number/J5677BAQOTZ6HQ3423WKSFWMF3/graph.json","events_json":"https://pith.science/api/pith-number/J5677BAQOTZ6HQ3423WKSFWMF3/events.json","paper":"https://pith.science/paper/J5677BAQ"},"agent_actions":{"view_html":"https://pith.science/pith/J5677BAQOTZ6HQ3423WKSFWMF3","download_json":"https://pith.science/pith/J5677BAQOTZ6HQ3423WKSFWMF3.json","view_paper":"https://pith.science/paper/J5677BAQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.07878&json=true","fetch_graph":"https://pith.science/api/pith-number/J5677BAQOTZ6HQ3423WKSFWMF3/graph.json","fetch_events":"https://pith.science/api/pith-number/J5677BAQOTZ6HQ3423WKSFWMF3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J5677BAQOTZ6HQ3423WKSFWMF3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J5677BAQOTZ6HQ3423WKSFWMF3/action/storage_attestation","attest_author":"https://pith.science/pith/J5677BAQOTZ6HQ3423WKSFWMF3/action/author_attestation","sign_citation":"https://pith.science/pith/J5677BAQOTZ6HQ3423WKSFWMF3/action/citation_signature","submit_replication":"https://pith.science/pith/J5677BAQOTZ6HQ3423WKSFWMF3/action/replication_record"}},"created_at":"2026-07-05T10:47:21.349085+00:00","updated_at":"2026-07-05T10:47:21.349085+00:00"}