{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JPZITHN46HE42HERDWPRG5AH5R","short_pith_number":"pith:JPZITHN4","schema_version":"1.0","canonical_sha256":"4bf2899dbcf1c9cd1c911d9f137407ec4bed5bf94af378130380c688136b4572","source":{"kind":"arxiv","id":"2501.11779","version":2},"attestation_state":"computed","paper":{"title":"Glinthawk: A Two-Tiered Architecture for Offline LLM Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC","cs.PF"],"primary_cat":"cs.LG","authors_text":"Pouya Hamadanian, Sadjad Fouladi","submitted_at":"2025-01-20T23:10:13Z","abstract_excerpt":"We introduce Glinthawk, an architecture for offline Large Language Model (LLM) inference. By leveraging a two-tiered structure, Glinthawk optimizes the utilization of the high-end accelerators (\"Tier 1\") by offloading the attention mechanism to lower-end compute tier (\"Tier 2\"). This separation allows the memory demand of the attention, known as the key-value cache, to scale independently from the model weights, enabling larger batch sizes and more efficient accelerator usage. Prototyped with NVIDIA T4 GPUs and standard CPU VMs, Glinthawk improves throughput by $5.9\\times$ and reduces cost of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.11779","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-01-20T23:10:13Z","cross_cats_sorted":["cs.DC","cs.PF"],"title_canon_sha256":"6fb070512dff6ba07dcf5bcab82209abb05a03e7b41ed2991ccd32bb63fed665","abstract_canon_sha256":"2b4022f9e9054ef0298abedfb1f0c25e0d33ad35cd7c600f5159f41a831469a6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:52.031756Z","signature_b64":"G6ZOc2OSYaHzwXU8fki2c/w6sipbyBw4KCQcmxKA+zNyqQFQg5LckiuAKeLg7SnK0YJY1YHNEC4kIqXlQXBpDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4bf2899dbcf1c9cd1c911d9f137407ec4bed5bf94af378130380c688136b4572","last_reissued_at":"2026-07-05T10:12:52.031237Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:52.031237Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Glinthawk: A Two-Tiered Architecture for Offline LLM Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC","cs.PF"],"primary_cat":"cs.LG","authors_text":"Pouya Hamadanian, Sadjad Fouladi","submitted_at":"2025-01-20T23:10:13Z","abstract_excerpt":"We introduce Glinthawk, an architecture for offline Large Language Model (LLM) inference. By leveraging a two-tiered structure, Glinthawk optimizes the utilization of the high-end accelerators (\"Tier 1\") by offloading the attention mechanism to lower-end compute tier (\"Tier 2\"). This separation allows the memory demand of the attention, known as the key-value cache, to scale independently from the model weights, enabling larger batch sizes and more efficient accelerator usage. Prototyped with NVIDIA T4 GPUs and standard CPU VMs, Glinthawk improves throughput by $5.9\\times$ and reduces cost of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.11779","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.11779/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.11779","created_at":"2026-07-05T10:12:52.031304+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.11779v2","created_at":"2026-07-05T10:12:52.031304+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.11779","created_at":"2026-07-05T10:12:52.031304+00:00"},{"alias_kind":"pith_short_12","alias_value":"JPZITHN46HE4","created_at":"2026-07-05T10:12:52.031304+00:00"},{"alias_kind":"pith_short_16","alias_value":"JPZITHN46HE42HER","created_at":"2026-07-05T10:12:52.031304+00:00"},{"alias_kind":"pith_short_8","alias_value":"JPZITHN4","created_at":"2026-07-05T10:12:52.031304+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.28095","citing_title":"SiDP: Memory-Efficient Data Parallelism for Offline LLM Inference","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JPZITHN46HE42HERDWPRG5AH5R","json":"https://pith.science/pith/JPZITHN46HE42HERDWPRG5AH5R.json","graph_json":"https://pith.science/api/pith-number/JPZITHN46HE42HERDWPRG5AH5R/graph.json","events_json":"https://pith.science/api/pith-number/JPZITHN46HE42HERDWPRG5AH5R/events.json","paper":"https://pith.science/paper/JPZITHN4"},"agent_actions":{"view_html":"https://pith.science/pith/JPZITHN46HE42HERDWPRG5AH5R","download_json":"https://pith.science/pith/JPZITHN46HE42HERDWPRG5AH5R.json","view_paper":"https://pith.science/paper/JPZITHN4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.11779&json=true","fetch_graph":"https://pith.science/api/pith-number/JPZITHN46HE42HERDWPRG5AH5R/graph.json","fetch_events":"https://pith.science/api/pith-number/JPZITHN46HE42HERDWPRG5AH5R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JPZITHN46HE42HERDWPRG5AH5R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JPZITHN46HE42HERDWPRG5AH5R/action/storage_attestation","attest_author":"https://pith.science/pith/JPZITHN46HE42HERDWPRG5AH5R/action/author_attestation","sign_citation":"https://pith.science/pith/JPZITHN46HE42HERDWPRG5AH5R/action/citation_signature","submit_replication":"https://pith.science/pith/JPZITHN46HE42HERDWPRG5AH5R/action/replication_record"}},"created_at":"2026-07-05T10:12:52.031304+00:00","updated_at":"2026-07-05T10:12:52.031304+00:00"}