{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5YGN37HHEG7GXKBJ722BEVVMIG","short_pith_number":"pith:5YGN37HH","schema_version":"1.0","canonical_sha256":"ee0cddfce721be6ba829feb41256ac41a3985b16de316f6f083e9fb625aae3ac","source":{"kind":"arxiv","id":"2406.14424","version":1},"attestation_state":"computed","paper":{"title":"CascadeServe: Unlocking Model Cascades for Inference Serving","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Alex Turk, Ferdi Kossmann, Lei Cao, Nesime Tatbul, Samuel Madden, Ziniu Wu","submitted_at":"2024-06-20T15:47:37Z","abstract_excerpt":"Machine learning (ML) models are increasingly deployed to production, calling for efficient inference serving systems. Efficient inference serving is complicated by two challenges: (i) ML models incur high computational costs, and (ii) the request arrival rates of practical applications have frequent, high, and sudden variations which make it hard to correctly provision hardware. Model cascades are positioned to tackle both of these challenges, as they (i) save work while maintaining accuracy, and (ii) expose a high-resolution trade-off between work and accuracy, allowing for fine-grained adju"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.14424","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2024-06-20T15:47:37Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"145b4d0fb4b331bc9e5c3e7c6b21c6e6eabf195651090e0fc1a6b76af02a8fea","abstract_canon_sha256":"78882cc906dcba1a7a3352bf54bf48b2acc4713ad261fb17e0d2067640faf10c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:34:50.605126Z","signature_b64":"hHJsou11fPdk3TkwDL4/4oAS2oUBEy7mifAuPWrNOif8ZAy0qsOExS3Q2xAhb5rYs49aULYjJXRckf9P6EZbAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ee0cddfce721be6ba829feb41256ac41a3985b16de316f6f083e9fb625aae3ac","last_reissued_at":"2026-07-05T08:34:50.604642Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:34:50.604642Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CascadeServe: Unlocking Model Cascades for Inference Serving","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Alex Turk, Ferdi Kossmann, Lei Cao, Nesime Tatbul, Samuel Madden, Ziniu Wu","submitted_at":"2024-06-20T15:47:37Z","abstract_excerpt":"Machine learning (ML) models are increasingly deployed to production, calling for efficient inference serving systems. Efficient inference serving is complicated by two challenges: (i) ML models incur high computational costs, and (ii) the request arrival rates of practical applications have frequent, high, and sudden variations which make it hard to correctly provision hardware. Model cascades are positioned to tackle both of these challenges, as they (i) save work while maintaining accuracy, and (ii) expose a high-resolution trade-off between work and accuracy, allowing for fine-grained adju"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.14424","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.14424/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.14424","created_at":"2026-07-05T08:34:50.604695+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.14424v1","created_at":"2026-07-05T08:34:50.604695+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.14424","created_at":"2026-07-05T08:34:50.604695+00:00"},{"alias_kind":"pith_short_12","alias_value":"5YGN37HHEG7G","created_at":"2026-07-05T08:34:50.604695+00:00"},{"alias_kind":"pith_short_16","alias_value":"5YGN37HHEG7GXKBJ","created_at":"2026-07-05T08:34:50.604695+00:00"},{"alias_kind":"pith_short_8","alias_value":"5YGN37HH","created_at":"2026-07-05T08:34:50.604695+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07923","citing_title":"Larch: Learned Query Optimization for Semantic Predicates","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5YGN37HHEG7GXKBJ722BEVVMIG","json":"https://pith.science/pith/5YGN37HHEG7GXKBJ722BEVVMIG.json","graph_json":"https://pith.science/api/pith-number/5YGN37HHEG7GXKBJ722BEVVMIG/graph.json","events_json":"https://pith.science/api/pith-number/5YGN37HHEG7GXKBJ722BEVVMIG/events.json","paper":"https://pith.science/paper/5YGN37HH"},"agent_actions":{"view_html":"https://pith.science/pith/5YGN37HHEG7GXKBJ722BEVVMIG","download_json":"https://pith.science/pith/5YGN37HHEG7GXKBJ722BEVVMIG.json","view_paper":"https://pith.science/paper/5YGN37HH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.14424&json=true","fetch_graph":"https://pith.science/api/pith-number/5YGN37HHEG7GXKBJ722BEVVMIG/graph.json","fetch_events":"https://pith.science/api/pith-number/5YGN37HHEG7GXKBJ722BEVVMIG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5YGN37HHEG7GXKBJ722BEVVMIG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5YGN37HHEG7GXKBJ722BEVVMIG/action/storage_attestation","attest_author":"https://pith.science/pith/5YGN37HHEG7GXKBJ722BEVVMIG/action/author_attestation","sign_citation":"https://pith.science/pith/5YGN37HHEG7GXKBJ722BEVVMIG/action/citation_signature","submit_replication":"https://pith.science/pith/5YGN37HHEG7GXKBJ722BEVVMIG/action/replication_record"}},"created_at":"2026-07-05T08:34:50.604695+00:00","updated_at":"2026-07-05T08:34:50.604695+00:00"}