{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:5WW6Y2ZTYZHI7QUOXFNHH2XWF6","short_pith_number":"pith:5WW6Y2ZT","schema_version":"1.0","canonical_sha256":"edadec6b33c64e8fc28eb95a73eaf62fbc728180651072e09709ac2bfe4e8425","source":{"kind":"arxiv","id":"2203.09040","version":3},"attestation_state":"computed","paper":{"title":"A Survey of Multi-Tenant Deep Learning Inference on GPU","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AR"],"primary_cat":"cs.DC","authors_text":"Chenchen Liu, Di Wang, Fuxun Yu, Longfei Shangguan, Minjia Zhang, Xiang Chen","submitted_at":"2022-03-17T02:37:29Z","abstract_excerpt":"Deep Learning (DL) models have achieved superior performance. Meanwhile, computing hardware like NVIDIA GPUs also demonstrated strong computing scaling trends with 2x throughput and memory bandwidth for each generation. With such strong computing scaling of GPUs, multi-tenant deep learning inference by co-locating multiple DL models onto the same GPU becomes widely deployed to improve resource utilization, enhance serving throughput, reduce energy cost, etc. However, achieving efficient multi-tenant DL inference is challenging which requires thorough full-stack system optimization. This survey"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.09040","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2022-03-17T02:37:29Z","cross_cats_sorted":["cs.AR"],"title_canon_sha256":"1937a1fe1649fcbf2e4b752a1048d3b9146cf16a6b32e8308532d72cd64cf528","abstract_canon_sha256":"6c5dc99532c76caa9cd865e41eefac5528d57ed57efcfa93f3831304bbe97c50"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:26:22.916705Z","signature_b64":"b0s1EpQsRjHm9E+QKK0hn7WVRg5PK+JFIlGBUkfclmX4/ECcYVxPWcEZt1VUxL9lw9JlRGbVp+lL2dFtfTqzCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"edadec6b33c64e8fc28eb95a73eaf62fbc728180651072e09709ac2bfe4e8425","last_reissued_at":"2026-07-05T04:26:22.916247Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:26:22.916247Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey of Multi-Tenant Deep Learning Inference on GPU","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AR"],"primary_cat":"cs.DC","authors_text":"Chenchen Liu, Di Wang, Fuxun Yu, Longfei Shangguan, Minjia Zhang, Xiang Chen","submitted_at":"2022-03-17T02:37:29Z","abstract_excerpt":"Deep Learning (DL) models have achieved superior performance. Meanwhile, computing hardware like NVIDIA GPUs also demonstrated strong computing scaling trends with 2x throughput and memory bandwidth for each generation. With such strong computing scaling of GPUs, multi-tenant deep learning inference by co-locating multiple DL models onto the same GPU becomes widely deployed to improve resource utilization, enhance serving throughput, reduce energy cost, etc. However, achieving efficient multi-tenant DL inference is challenging which requires thorough full-stack system optimization. This survey"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.09040","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.09040/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.09040","created_at":"2026-07-05T04:26:22.916299+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.09040v3","created_at":"2026-07-05T04:26:22.916299+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.09040","created_at":"2026-07-05T04:26:22.916299+00:00"},{"alias_kind":"pith_short_12","alias_value":"5WW6Y2ZTYZHI","created_at":"2026-07-05T04:26:22.916299+00:00"},{"alias_kind":"pith_short_16","alias_value":"5WW6Y2ZTYZHI7QUO","created_at":"2026-07-05T04:26:22.916299+00:00"},{"alias_kind":"pith_short_8","alias_value":"5WW6Y2ZT","created_at":"2026-07-05T04:26:22.916299+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2404.00507","citing_title":"THEMIS: Time, Heterogeneity, and Energy Minded Scheduling for Fair Multi-Tenant Use in FPGAs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28175","citing_title":"Strait: Perceiving Priority and Interference in ML Inference Serving","ref_index":108,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5WW6Y2ZTYZHI7QUOXFNHH2XWF6","json":"https://pith.science/pith/5WW6Y2ZTYZHI7QUOXFNHH2XWF6.json","graph_json":"https://pith.science/api/pith-number/5WW6Y2ZTYZHI7QUOXFNHH2XWF6/graph.json","events_json":"https://pith.science/api/pith-number/5WW6Y2ZTYZHI7QUOXFNHH2XWF6/events.json","paper":"https://pith.science/paper/5WW6Y2ZT"},"agent_actions":{"view_html":"https://pith.science/pith/5WW6Y2ZTYZHI7QUOXFNHH2XWF6","download_json":"https://pith.science/pith/5WW6Y2ZTYZHI7QUOXFNHH2XWF6.json","view_paper":"https://pith.science/paper/5WW6Y2ZT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.09040&json=true","fetch_graph":"https://pith.science/api/pith-number/5WW6Y2ZTYZHI7QUOXFNHH2XWF6/graph.json","fetch_events":"https://pith.science/api/pith-number/5WW6Y2ZTYZHI7QUOXFNHH2XWF6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5WW6Y2ZTYZHI7QUOXFNHH2XWF6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5WW6Y2ZTYZHI7QUOXFNHH2XWF6/action/storage_attestation","attest_author":"https://pith.science/pith/5WW6Y2ZTYZHI7QUOXFNHH2XWF6/action/author_attestation","sign_citation":"https://pith.science/pith/5WW6Y2ZTYZHI7QUOXFNHH2XWF6/action/citation_signature","submit_replication":"https://pith.science/pith/5WW6Y2ZTYZHI7QUOXFNHH2XWF6/action/replication_record"}},"created_at":"2026-07-05T04:26:22.916299+00:00","updated_at":"2026-07-05T04:26:22.916299+00:00"}