{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:EHHJYOH64ZSTS4KSNPCNFHLDDS","short_pith_number":"pith:EHHJYOH6","schema_version":"1.0","canonical_sha256":"21ce9c38fee6653971526bc4d29d631ca305e3a5d97b84da2a81e0e7c3c6d56c","source":{"kind":"arxiv","id":"2607.10186","version":1},"attestation_state":"computed","paper":{"title":"FlashAccel: Leveraging High-Bandwidth Flash for High-Throughput LLM Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AR","authors_text":"Chunmeng Dou, Xiaoming Chen, Xiaotian Sun, Xiaoyu Zhang, Xinyu Wang, Xueqi Li, Yalong Xue","submitted_at":"2026-07-11T08:03:13Z","abstract_excerpt":"Large language model (LLM) inference is increasingly limited by the capacity of High-Bandwidth Memory (HBM) in GPUs, as model weights and KV cache grow rapidly. High-Bandwidth Flash (HBF) provides higher capacity than HBM while retaining comparable bandwidth, making it a promising substrate for capacity-constrained LLM inference. However, its inherently high access latency, low bandwidth utilization, and lack of support for heterogeneous resource management make it difficult to integrate HBF into GPUs for LLM inference. We present FlashAccel, a co-designed system that enables efficient LLM inf"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.10186","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AR","submitted_at":"2026-07-11T08:03:13Z","cross_cats_sorted":[],"title_canon_sha256":"a56a3973aa222b5f00c71a5dd96114e3db4906d53d8c1dd11629e10a210e485c","abstract_canon_sha256":"68c49d0b0033d5842e0db1e72c3d7934658cc037d116926b9c8ba5a5fff8c098"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-14T01:20:29.642208Z","signature_b64":"MHmeaBiK78j6GLaHNoF6mSzzalkuoWCYPA3TiF91e/iiVdZt0VmcOLH5VU9tWCKWDJPeG7DtxBxt4z+k6MSyAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"21ce9c38fee6653971526bc4d29d631ca305e3a5d97b84da2a81e0e7c3c6d56c","last_reissued_at":"2026-07-14T01:20:29.641380Z","signature_status":"signed_v1","first_computed_at":"2026-07-14T01:20:29.641380Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FlashAccel: Leveraging High-Bandwidth Flash for High-Throughput LLM Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AR","authors_text":"Chunmeng Dou, Xiaoming Chen, Xiaotian Sun, Xiaoyu Zhang, Xinyu Wang, Xueqi Li, Yalong Xue","submitted_at":"2026-07-11T08:03:13Z","abstract_excerpt":"Large language model (LLM) inference is increasingly limited by the capacity of High-Bandwidth Memory (HBM) in GPUs, as model weights and KV cache grow rapidly. High-Bandwidth Flash (HBF) provides higher capacity than HBM while retaining comparable bandwidth, making it a promising substrate for capacity-constrained LLM inference. However, its inherently high access latency, low bandwidth utilization, and lack of support for heterogeneous resource management make it difficult to integrate HBF into GPUs for LLM inference. We present FlashAccel, a co-designed system that enables efficient LLM inf"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.10186","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.10186/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.10186","created_at":"2026-07-14T01:20:29.641817+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.10186v1","created_at":"2026-07-14T01:20:29.641817+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.10186","created_at":"2026-07-14T01:20:29.641817+00:00"},{"alias_kind":"pith_short_12","alias_value":"EHHJYOH64ZST","created_at":"2026-07-14T01:20:29.641817+00:00"},{"alias_kind":"pith_short_16","alias_value":"EHHJYOH64ZSTS4KS","created_at":"2026-07-14T01:20:29.641817+00:00"},{"alias_kind":"pith_short_8","alias_value":"EHHJYOH6","created_at":"2026-07-14T01:20:29.641817+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EHHJYOH64ZSTS4KSNPCNFHLDDS","json":"https://pith.science/pith/EHHJYOH64ZSTS4KSNPCNFHLDDS.json","graph_json":"https://pith.science/api/pith-number/EHHJYOH64ZSTS4KSNPCNFHLDDS/graph.json","events_json":"https://pith.science/api/pith-number/EHHJYOH64ZSTS4KSNPCNFHLDDS/events.json","paper":"https://pith.science/paper/EHHJYOH6"},"agent_actions":{"view_html":"https://pith.science/pith/EHHJYOH64ZSTS4KSNPCNFHLDDS","download_json":"https://pith.science/pith/EHHJYOH64ZSTS4KSNPCNFHLDDS.json","view_paper":"https://pith.science/paper/EHHJYOH6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.10186&json=true","fetch_graph":"https://pith.science/api/pith-number/EHHJYOH64ZSTS4KSNPCNFHLDDS/graph.json","fetch_events":"https://pith.science/api/pith-number/EHHJYOH64ZSTS4KSNPCNFHLDDS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EHHJYOH64ZSTS4KSNPCNFHLDDS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EHHJYOH64ZSTS4KSNPCNFHLDDS/action/storage_attestation","attest_author":"https://pith.science/pith/EHHJYOH64ZSTS4KSNPCNFHLDDS/action/author_attestation","sign_citation":"https://pith.science/pith/EHHJYOH64ZSTS4KSNPCNFHLDDS/action/citation_signature","submit_replication":"https://pith.science/pith/EHHJYOH64ZSTS4KSNPCNFHLDDS/action/replication_record"}},"created_at":"2026-07-14T01:20:29.641817+00:00","updated_at":"2026-07-14T01:20:29.641817+00:00"}