{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:O72FLWIUGKWY4ZCNOOUZTQ7JIC","short_pith_number":"pith:O72FLWIU","schema_version":"1.0","canonical_sha256":"77f455d91432ad8e644d73a999c3e940a82ddeaa95ded69179586cd5a7d88db1","source":{"kind":"arxiv","id":"2607.20327","version":1},"attestation_state":"computed","paper":{"title":"PyroDash: Cost-Efficient Token-Level Small-Large Language Model Collaborative Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jianlin Zhong, Jianyao Ma, Niqi Lyu, Pengtao Shi, Sicong Xia, Wei Qiu, Yicheng Ding","submitted_at":"2026-07-22T16:14:26Z","abstract_excerpt":"Large language models (LLMs) provide strong reasoning capabilities but are expensive to serve at scale, whereas small language models (SLMs) are cheaper but less reliable on difficult problems. We introduce PyroDash, a cost-aware framework for token-level SLM-LLM collaborative inference. During generation, the SLM decides whether to request assistance by emitting a control token. A Collaborate Engine then sends the query and partial reasoning trace to a frozen LLM for completion through a single handoff. The policy is internalized in the SLM, requiring neither a separate router, LLM retraining"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.20327","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-07-22T16:14:26Z","cross_cats_sorted":[],"title_canon_sha256":"c157b6885b9ce9b98fea39a5ce6a263d26b27d720d5b0249ddb7a04cda43c33d","abstract_canon_sha256":"b5c3fc502dd50edf1cc493f408f95afbf2fae2db9f6a90306ed1f74af2bb71b5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-23T01:25:14.620631Z","signature_b64":"ixieJBKklGFLPt5B7rbmrsFSKPhsg5FXDdG4hC105BtjRa6FAbX4hCWAZA4q9Gotud0vQA3C+gdtW18o02dECQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77f455d91432ad8e644d73a999c3e940a82ddeaa95ded69179586cd5a7d88db1","last_reissued_at":"2026-07-23T01:25:14.619834Z","signature_status":"signed_v1","first_computed_at":"2026-07-23T01:25:14.619834Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PyroDash: Cost-Efficient Token-Level Small-Large Language Model Collaborative Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jianlin Zhong, Jianyao Ma, Niqi Lyu, Pengtao Shi, Sicong Xia, Wei Qiu, Yicheng Ding","submitted_at":"2026-07-22T16:14:26Z","abstract_excerpt":"Large language models (LLMs) provide strong reasoning capabilities but are expensive to serve at scale, whereas small language models (SLMs) are cheaper but less reliable on difficult problems. We introduce PyroDash, a cost-aware framework for token-level SLM-LLM collaborative inference. During generation, the SLM decides whether to request assistance by emitting a control token. A Collaborate Engine then sends the query and partial reasoning trace to a frozen LLM for completion through a single handoff. The policy is internalized in the SLM, requiring neither a separate router, LLM retraining"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.20327","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.20327/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.20327","created_at":"2026-07-23T01:25:14.620243+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.20327v1","created_at":"2026-07-23T01:25:14.620243+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.20327","created_at":"2026-07-23T01:25:14.620243+00:00"},{"alias_kind":"pith_short_12","alias_value":"O72FLWIUGKWY","created_at":"2026-07-23T01:25:14.620243+00:00"},{"alias_kind":"pith_short_16","alias_value":"O72FLWIUGKWY4ZCN","created_at":"2026-07-23T01:25:14.620243+00:00"},{"alias_kind":"pith_short_8","alias_value":"O72FLWIU","created_at":"2026-07-23T01:25:14.620243+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O72FLWIUGKWY4ZCNOOUZTQ7JIC","json":"https://pith.science/pith/O72FLWIUGKWY4ZCNOOUZTQ7JIC.json","graph_json":"https://pith.science/api/pith-number/O72FLWIUGKWY4ZCNOOUZTQ7JIC/graph.json","events_json":"https://pith.science/api/pith-number/O72FLWIUGKWY4ZCNOOUZTQ7JIC/events.json","paper":"https://pith.science/paper/O72FLWIU"},"agent_actions":{"view_html":"https://pith.science/pith/O72FLWIUGKWY4ZCNOOUZTQ7JIC","download_json":"https://pith.science/pith/O72FLWIUGKWY4ZCNOOUZTQ7JIC.json","view_paper":"https://pith.science/paper/O72FLWIU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.20327&json=true","fetch_graph":"https://pith.science/api/pith-number/O72FLWIUGKWY4ZCNOOUZTQ7JIC/graph.json","fetch_events":"https://pith.science/api/pith-number/O72FLWIUGKWY4ZCNOOUZTQ7JIC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O72FLWIUGKWY4ZCNOOUZTQ7JIC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O72FLWIUGKWY4ZCNOOUZTQ7JIC/action/storage_attestation","attest_author":"https://pith.science/pith/O72FLWIUGKWY4ZCNOOUZTQ7JIC/action/author_attestation","sign_citation":"https://pith.science/pith/O72FLWIUGKWY4ZCNOOUZTQ7JIC/action/citation_signature","submit_replication":"https://pith.science/pith/O72FLWIUGKWY4ZCNOOUZTQ7JIC/action/replication_record"}},"created_at":"2026-07-23T01:25:14.620243+00:00","updated_at":"2026-07-23T01:25:14.620243+00:00"}