{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FPFCJ4IQGQGRPMFG2PZIGOIES2","short_pith_number":"pith:FPFCJ4IQ","schema_version":"1.0","canonical_sha256":"2bca24f110340d17b0a6d3f283390496b39a92971a35444a5f445f0b8455eb92","source":{"kind":"arxiv","id":"2505.03763","version":1},"attestation_state":"computed","paper":{"title":"Splitwiser: Efficient LM inference with constrained resources","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.DC","cs.LG"],"primary_cat":"cs.AR","authors_text":"Adney Cardoza, Asad Aali, Melissa Capo","submitted_at":"2025-04-21T00:21:08Z","abstract_excerpt":"Efficient inference of LLMs remains a crucial challenge, with two main phases: a compute-intensive prompt computation and a memory-intensive token generation. Despite existing batching and scheduling techniques, token generation phases fail to fully utilize compute resources, especially when compared to prompt computation phases. To address these challenges, we propose Splitwiser, a methodology that splits the two phases of an LLM inference request onto the same GPU, thereby reducing overhead and improving memory access and cache utilization. By eliminating the need to transfer data across dev"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.03763","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AR","submitted_at":"2025-04-21T00:21:08Z","cross_cats_sorted":["cs.AI","cs.DC","cs.LG"],"title_canon_sha256":"0b38b4ec3fb0ffb9f63ae3fc2d9bdd2e59d4992717208f3cd5f8aad75c88a30f","abstract_canon_sha256":"417bf4e91f28679e2499796425712a5fb03254aa27bce8d585176b80a0156e25"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:59:30.607915Z","signature_b64":"/7GQsZg0S8o35FSfMuukQXsA0xFZJAwa9XSskSMhVwsVTrODnyHXiDM+5mo+Y5uM/d3XNj0bVipEEFfcpKdDBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2bca24f110340d17b0a6d3f283390496b39a92971a35444a5f445f0b8455eb92","last_reissued_at":"2026-07-05T10:59:30.607413Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:59:30.607413Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Splitwiser: Efficient LM inference with constrained resources","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.DC","cs.LG"],"primary_cat":"cs.AR","authors_text":"Adney Cardoza, Asad Aali, Melissa Capo","submitted_at":"2025-04-21T00:21:08Z","abstract_excerpt":"Efficient inference of LLMs remains a crucial challenge, with two main phases: a compute-intensive prompt computation and a memory-intensive token generation. Despite existing batching and scheduling techniques, token generation phases fail to fully utilize compute resources, especially when compared to prompt computation phases. To address these challenges, we propose Splitwiser, a methodology that splits the two phases of an LLM inference request onto the same GPU, thereby reducing overhead and improving memory access and cache utilization. By eliminating the need to transfer data across dev"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.03763","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.03763/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.03763","created_at":"2026-07-05T10:59:30.607480+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.03763v1","created_at":"2026-07-05T10:59:30.607480+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.03763","created_at":"2026-07-05T10:59:30.607480+00:00"},{"alias_kind":"pith_short_12","alias_value":"FPFCJ4IQGQGR","created_at":"2026-07-05T10:59:30.607480+00:00"},{"alias_kind":"pith_short_16","alias_value":"FPFCJ4IQGQGRPMFG","created_at":"2026-07-05T10:59:30.607480+00:00"},{"alias_kind":"pith_short_8","alias_value":"FPFCJ4IQ","created_at":"2026-07-05T10:59:30.607480+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.22906","citing_title":"Network Edge Inference for Large Language Models: Principles, Techniques, and Opportunities","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FPFCJ4IQGQGRPMFG2PZIGOIES2","json":"https://pith.science/pith/FPFCJ4IQGQGRPMFG2PZIGOIES2.json","graph_json":"https://pith.science/api/pith-number/FPFCJ4IQGQGRPMFG2PZIGOIES2/graph.json","events_json":"https://pith.science/api/pith-number/FPFCJ4IQGQGRPMFG2PZIGOIES2/events.json","paper":"https://pith.science/paper/FPFCJ4IQ"},"agent_actions":{"view_html":"https://pith.science/pith/FPFCJ4IQGQGRPMFG2PZIGOIES2","download_json":"https://pith.science/pith/FPFCJ4IQGQGRPMFG2PZIGOIES2.json","view_paper":"https://pith.science/paper/FPFCJ4IQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.03763&json=true","fetch_graph":"https://pith.science/api/pith-number/FPFCJ4IQGQGRPMFG2PZIGOIES2/graph.json","fetch_events":"https://pith.science/api/pith-number/FPFCJ4IQGQGRPMFG2PZIGOIES2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FPFCJ4IQGQGRPMFG2PZIGOIES2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FPFCJ4IQGQGRPMFG2PZIGOIES2/action/storage_attestation","attest_author":"https://pith.science/pith/FPFCJ4IQGQGRPMFG2PZIGOIES2/action/author_attestation","sign_citation":"https://pith.science/pith/FPFCJ4IQGQGRPMFG2PZIGOIES2/action/citation_signature","submit_replication":"https://pith.science/pith/FPFCJ4IQGQGRPMFG2PZIGOIES2/action/replication_record"}},"created_at":"2026-07-05T10:59:30.607480+00:00","updated_at":"2026-07-05T10:59:30.607480+00:00"}