{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OLZKVGIJJY7EKKVTO2FKEPPXVD","short_pith_number":"pith:OLZKVGIJ","schema_version":"1.0","canonical_sha256":"72f2aa99094e3e452ab3768aa23df7a8c679e4b20b6c51e93f8950bc5ef29d97","source":{"kind":"arxiv","id":"2301.12444","version":1},"attestation_state":"computed","paper":{"title":"Exploring Attention Map Reuse for Efficient Transformer Neural Networks","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["eess.SP"],"primary_cat":"cs.AI","authors_text":"Jungwook Choi, Kyuhong Shim, Wonyong Sung","submitted_at":"2023-01-29T13:38:45Z","abstract_excerpt":"Transformer-based deep neural networks have achieved great success in various sequence applications due to their powerful ability to model long-range dependency. The key module of Transformer is self-attention (SA) which extracts features from the entire sequence regardless of the distance between positions. Although SA helps Transformer performs particularly well on long-range tasks, SA requires quadratic computation and memory complexity with the input sequence length. Recently, attention map reuse, which groups multiple SA layers to share one attention map, has been proposed and achieved si"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.12444","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.AI","submitted_at":"2023-01-29T13:38:45Z","cross_cats_sorted":["eess.SP"],"title_canon_sha256":"bea94cbf808ab87347450670485bcb3bad747ea8cc6c8166c4a611bacadf222c","abstract_canon_sha256":"02a2e4fba994e62ed6b3c68cc488c529dd635c0a4467b9b6ecc7c6166f54ea1f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:36:54.426988Z","signature_b64":"aDe70jkdlo0H/HA0itOxVesesEYY2ABDzBWukbIfJKZ59YhvfoDROZZBL5Xxw+IXgrunQA75i1JdhFe5ec+ZDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"72f2aa99094e3e452ab3768aa23df7a8c679e4b20b6c51e93f8950bc5ef29d97","last_reissued_at":"2026-07-05T05:36:54.426596Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:36:54.426596Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring Attention Map Reuse for Efficient Transformer Neural Networks","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["eess.SP"],"primary_cat":"cs.AI","authors_text":"Jungwook Choi, Kyuhong Shim, Wonyong Sung","submitted_at":"2023-01-29T13:38:45Z","abstract_excerpt":"Transformer-based deep neural networks have achieved great success in various sequence applications due to their powerful ability to model long-range dependency. The key module of Transformer is self-attention (SA) which extracts features from the entire sequence regardless of the distance between positions. Although SA helps Transformer performs particularly well on long-range tasks, SA requires quadratic computation and memory complexity with the input sequence length. Recently, attention map reuse, which groups multiple SA layers to share one attention map, has been proposed and achieved si"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.12444","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.12444/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.12444","created_at":"2026-07-05T05:36:54.426654+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.12444v1","created_at":"2026-07-05T05:36:54.426654+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.12444","created_at":"2026-07-05T05:36:54.426654+00:00"},{"alias_kind":"pith_short_12","alias_value":"OLZKVGIJJY7E","created_at":"2026-07-05T05:36:54.426654+00:00"},{"alias_kind":"pith_short_16","alias_value":"OLZKVGIJJY7EKKVT","created_at":"2026-07-05T05:36:54.426654+00:00"},{"alias_kind":"pith_short_8","alias_value":"OLZKVGIJ","created_at":"2026-07-05T05:36:54.426654+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.02344","citing_title":"UniForm: A Reuse Attention Mechanism Optimized for Efficient Vision Transformers on Edge Devices","ref_index":34,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OLZKVGIJJY7EKKVTO2FKEPPXVD","json":"https://pith.science/pith/OLZKVGIJJY7EKKVTO2FKEPPXVD.json","graph_json":"https://pith.science/api/pith-number/OLZKVGIJJY7EKKVTO2FKEPPXVD/graph.json","events_json":"https://pith.science/api/pith-number/OLZKVGIJJY7EKKVTO2FKEPPXVD/events.json","paper":"https://pith.science/paper/OLZKVGIJ"},"agent_actions":{"view_html":"https://pith.science/pith/OLZKVGIJJY7EKKVTO2FKEPPXVD","download_json":"https://pith.science/pith/OLZKVGIJJY7EKKVTO2FKEPPXVD.json","view_paper":"https://pith.science/paper/OLZKVGIJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.12444&json=true","fetch_graph":"https://pith.science/api/pith-number/OLZKVGIJJY7EKKVTO2FKEPPXVD/graph.json","fetch_events":"https://pith.science/api/pith-number/OLZKVGIJJY7EKKVTO2FKEPPXVD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OLZKVGIJJY7EKKVTO2FKEPPXVD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OLZKVGIJJY7EKKVTO2FKEPPXVD/action/storage_attestation","attest_author":"https://pith.science/pith/OLZKVGIJJY7EKKVTO2FKEPPXVD/action/author_attestation","sign_citation":"https://pith.science/pith/OLZKVGIJJY7EKKVTO2FKEPPXVD/action/citation_signature","submit_replication":"https://pith.science/pith/OLZKVGIJJY7EKKVTO2FKEPPXVD/action/replication_record"}},"created_at":"2026-07-05T05:36:54.426654+00:00","updated_at":"2026-07-05T05:36:54.426654+00:00"}