{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DPOBLJWUMV4C424FWBUJ54LJYV","short_pith_number":"pith:DPOBLJWU","schema_version":"1.0","canonical_sha256":"1bdc15a6d465782e6b85b0689ef169c56168b05113c11397d2fc3173aeb7b108","source":{"kind":"arxiv","id":"2407.12866","version":1},"attestation_state":"computed","paper":{"title":"Beyond KV Caching: Shared Attention for Efficient LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bingli Liao, Danilo Vasconcellos Vargas","submitted_at":"2024-07-13T07:23:07Z","abstract_excerpt":"The efficiency of large language models (LLMs) remains a critical challenge, particularly in contexts where computational resources are limited. Traditional attention mechanisms in these models, while powerful, require significant computational and memory resources due to the necessity of recalculating and storing attention weights across different layers. This paper introduces a novel Shared Attention (SA) mechanism, designed to enhance the efficiency of LLMs by directly sharing computed attention weights across multiple layers. Unlike previous methods that focus on sharing intermediate Key-V"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.12866","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-13T07:23:07Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"a91a69af96aaa289cecf6d37186e54a93567912338cee2a2c225498b1885417b","abstract_canon_sha256":"10c8b584ebef2e6008f5d2ea82a56f9dc69b90c83773e3c7c1461f2307e86125"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:12.806511Z","signature_b64":"u2fuUVfj4qglYwvQIsROZBKGFZwxrj3uuip5BSRiALO9snvFZn0t2XpcAa9ZmYhFdDpdla0oOWqm6zIACv4QDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1bdc15a6d465782e6b85b0689ef169c56168b05113c11397d2fc3173aeb7b108","last_reissued_at":"2026-07-05T08:45:12.806003Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:12.806003Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond KV Caching: Shared Attention for Efficient LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bingli Liao, Danilo Vasconcellos Vargas","submitted_at":"2024-07-13T07:23:07Z","abstract_excerpt":"The efficiency of large language models (LLMs) remains a critical challenge, particularly in contexts where computational resources are limited. Traditional attention mechanisms in these models, while powerful, require significant computational and memory resources due to the necessity of recalculating and storing attention weights across different layers. This paper introduces a novel Shared Attention (SA) mechanism, designed to enhance the efficiency of LLMs by directly sharing computed attention weights across multiple layers. Unlike previous methods that focus on sharing intermediate Key-V"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.12866","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.12866/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.12866","created_at":"2026-07-05T08:45:12.806062+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.12866v1","created_at":"2026-07-05T08:45:12.806062+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.12866","created_at":"2026-07-05T08:45:12.806062+00:00"},{"alias_kind":"pith_short_12","alias_value":"DPOBLJWUMV4C","created_at":"2026-07-05T08:45:12.806062+00:00"},{"alias_kind":"pith_short_16","alias_value":"DPOBLJWUMV4C424F","created_at":"2026-07-05T08:45:12.806062+00:00"},{"alias_kind":"pith_short_8","alias_value":"DPOBLJWU","created_at":"2026-07-05T08:45:12.806062+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.17331","citing_title":"ECHO-LLaMA: Efficient Caching for High-Performance LLaMA Training","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DPOBLJWUMV4C424FWBUJ54LJYV","json":"https://pith.science/pith/DPOBLJWUMV4C424FWBUJ54LJYV.json","graph_json":"https://pith.science/api/pith-number/DPOBLJWUMV4C424FWBUJ54LJYV/graph.json","events_json":"https://pith.science/api/pith-number/DPOBLJWUMV4C424FWBUJ54LJYV/events.json","paper":"https://pith.science/paper/DPOBLJWU"},"agent_actions":{"view_html":"https://pith.science/pith/DPOBLJWUMV4C424FWBUJ54LJYV","download_json":"https://pith.science/pith/DPOBLJWUMV4C424FWBUJ54LJYV.json","view_paper":"https://pith.science/paper/DPOBLJWU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.12866&json=true","fetch_graph":"https://pith.science/api/pith-number/DPOBLJWUMV4C424FWBUJ54LJYV/graph.json","fetch_events":"https://pith.science/api/pith-number/DPOBLJWUMV4C424FWBUJ54LJYV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DPOBLJWUMV4C424FWBUJ54LJYV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DPOBLJWUMV4C424FWBUJ54LJYV/action/storage_attestation","attest_author":"https://pith.science/pith/DPOBLJWUMV4C424FWBUJ54LJYV/action/author_attestation","sign_citation":"https://pith.science/pith/DPOBLJWUMV4C424FWBUJ54LJYV/action/citation_signature","submit_replication":"https://pith.science/pith/DPOBLJWUMV4C424FWBUJ54LJYV/action/replication_record"}},"created_at":"2026-07-05T08:45:12.806062+00:00","updated_at":"2026-07-05T08:45:12.806062+00:00"}