{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XNSUUD3VGCCXBTU66PQG2R2UPW","short_pith_number":"pith:XNSUUD3V","schema_version":"1.0","canonical_sha256":"bb654a0f75308570ce9ef3e06d47547db1325dd8ed3f6799abb7ad831a0bf3e7","source":{"kind":"arxiv","id":"2506.02572","version":1},"attestation_state":"computed","paper":{"title":"HATA: Trainable and Hardware-Efficient Hash-Aware Top-k Attention for Scalable Large Model Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bowen Ye, Cheng Li, Feng Wu, Gong Zhang, Guanbin Xu, Jiawei Yi, Juncheng Zhang, Kun Yuan, Ouxiang Zhou, Ping Gong, Renhai Chen, Ruibo Liu, Shengnan Wang, Tong Yang, Youhui Bai, Zewen Jin","submitted_at":"2025-06-03T07:53:32Z","abstract_excerpt":"Large Language Models (LLMs) have emerged as a pivotal research area, yet the attention module remains a critical bottleneck in LLM inference, even with techniques like KVCache to mitigate redundant computations. While various top-$k$ attention mechanisms have been proposed to accelerate LLM inference by exploiting the inherent sparsity of attention, they often struggled to strike a balance between efficiency and accuracy. In this paper, we introduce HATA (Hash-Aware Top-$k$ Attention), a novel approach that systematically integrates low-overhead learning-to-hash techniques into the Top-$k$ at"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.02572","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-06-03T07:53:32Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fe41db6f064e5de78a41068bc380676d904e7ac1d782c5043ea58a67a8ab5f38","abstract_canon_sha256":"080f56f3b906c9c380cf5463c4483406d24e72afaae538ed5f204cb9dde77b0e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:04.519746Z","signature_b64":"XYg07MNACVc4fWwMs2EYE4Zy+lfIGpWOX+beuKK++iaJuH5efIShBGmnZzw4HZE0TTWvdzlGCetm3I02qaKiBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb654a0f75308570ce9ef3e06d47547db1325dd8ed3f6799abb7ad831a0bf3e7","last_reissued_at":"2026-07-05T11:15:04.519214Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:04.519214Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HATA: Trainable and Hardware-Efficient Hash-Aware Top-k Attention for Scalable Large Model Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bowen Ye, Cheng Li, Feng Wu, Gong Zhang, Guanbin Xu, Jiawei Yi, Juncheng Zhang, Kun Yuan, Ouxiang Zhou, Ping Gong, Renhai Chen, Ruibo Liu, Shengnan Wang, Tong Yang, Youhui Bai, Zewen Jin","submitted_at":"2025-06-03T07:53:32Z","abstract_excerpt":"Large Language Models (LLMs) have emerged as a pivotal research area, yet the attention module remains a critical bottleneck in LLM inference, even with techniques like KVCache to mitigate redundant computations. While various top-$k$ attention mechanisms have been proposed to accelerate LLM inference by exploiting the inherent sparsity of attention, they often struggled to strike a balance between efficiency and accuracy. In this paper, we introduce HATA (Hash-Aware Top-$k$ Attention), a novel approach that systematically integrates low-overhead learning-to-hash techniques into the Top-$k$ at"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.02572","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.02572/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.02572","created_at":"2026-07-05T11:15:04.519287+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.02572v1","created_at":"2026-07-05T11:15:04.519287+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.02572","created_at":"2026-07-05T11:15:04.519287+00:00"},{"alias_kind":"pith_short_12","alias_value":"XNSUUD3VGCCX","created_at":"2026-07-05T11:15:04.519287+00:00"},{"alias_kind":"pith_short_16","alias_value":"XNSUUD3VGCCXBTU6","created_at":"2026-07-05T11:15:04.519287+00:00"},{"alias_kind":"pith_short_8","alias_value":"XNSUUD3V","created_at":"2026-07-05T11:15:04.519287+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.04405","citing_title":"Training-Free Hashing-Based Attention via Binary Principal Components","ref_index":51,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XNSUUD3VGCCXBTU66PQG2R2UPW","json":"https://pith.science/pith/XNSUUD3VGCCXBTU66PQG2R2UPW.json","graph_json":"https://pith.science/api/pith-number/XNSUUD3VGCCXBTU66PQG2R2UPW/graph.json","events_json":"https://pith.science/api/pith-number/XNSUUD3VGCCXBTU66PQG2R2UPW/events.json","paper":"https://pith.science/paper/XNSUUD3V"},"agent_actions":{"view_html":"https://pith.science/pith/XNSUUD3VGCCXBTU66PQG2R2UPW","download_json":"https://pith.science/pith/XNSUUD3VGCCXBTU66PQG2R2UPW.json","view_paper":"https://pith.science/paper/XNSUUD3V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.02572&json=true","fetch_graph":"https://pith.science/api/pith-number/XNSUUD3VGCCXBTU66PQG2R2UPW/graph.json","fetch_events":"https://pith.science/api/pith-number/XNSUUD3VGCCXBTU66PQG2R2UPW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XNSUUD3VGCCXBTU66PQG2R2UPW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XNSUUD3VGCCXBTU66PQG2R2UPW/action/storage_attestation","attest_author":"https://pith.science/pith/XNSUUD3VGCCXBTU66PQG2R2UPW/action/author_attestation","sign_citation":"https://pith.science/pith/XNSUUD3VGCCXBTU66PQG2R2UPW/action/citation_signature","submit_replication":"https://pith.science/pith/XNSUUD3VGCCXBTU66PQG2R2UPW/action/replication_record"}},"created_at":"2026-07-05T11:15:04.519287+00:00","updated_at":"2026-07-05T11:15:04.519287+00:00"}