{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PD3MJVH5K2KHR4JPUGHERPLYD4","short_pith_number":"pith:PD3MJVH5","schema_version":"1.0","canonical_sha256":"78f6c4d4fd569478f12fa18e48bd781f32d35aea60d87021a8ecac9ec65919f1","source":{"kind":"arxiv","id":"2407.20485","version":2},"attestation_state":"computed","paper":{"title":"A2SF: Accumulative Attention Scoring with Forgetting Factor for Token Pruning in Transformer Decoder","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Dongkun Shin, Hyun-rae Jo","submitted_at":"2024-07-30T01:13:42Z","abstract_excerpt":"Recently, large language models (LLM) based on transformers are facing memory bottleneck issues due to KV cache, especially in long sequence handling. Previous researches proposed KV cache compression techniques that identify insignificant tokens based on Accumulative Attention Scores and removes their items from KV cache, noting that only few tokens play an important role in attention operations. However, we have observed that the existing Accumulative Attention Score is not suitable for the transformer decoder structure. In the decoder model, the number of times the Attention Score accumulat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.20485","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-30T01:13:42Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"6a3666d065bcdb8c7c5775f187d4af556ce77f7e5855361bd1debcfdbd8186ae","abstract_canon_sha256":"bed4e8abf7a21a72ec95ee44cd08b19adcb2ca1b2ab697a870b5fe3c6c51bdee"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:50:31.797663Z","signature_b64":"ZaIjQy7Dq0keP1YsqBZXljsI3bnAYMC8VcRLMD6xZMkEawcwOsyNgvpefEt6HgIwfNZ9zM5NtxLkwbapje33AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"78f6c4d4fd569478f12fa18e48bd781f32d35aea60d87021a8ecac9ec65919f1","last_reissued_at":"2026-07-05T08:50:31.797134Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:50:31.797134Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A2SF: Accumulative Attention Scoring with Forgetting Factor for Token Pruning in Transformer Decoder","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Dongkun Shin, Hyun-rae Jo","submitted_at":"2024-07-30T01:13:42Z","abstract_excerpt":"Recently, large language models (LLM) based on transformers are facing memory bottleneck issues due to KV cache, especially in long sequence handling. Previous researches proposed KV cache compression techniques that identify insignificant tokens based on Accumulative Attention Scores and removes their items from KV cache, noting that only few tokens play an important role in attention operations. However, we have observed that the existing Accumulative Attention Score is not suitable for the transformer decoder structure. In the decoder model, the number of times the Attention Score accumulat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.20485","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.20485/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.20485","created_at":"2026-07-05T08:50:31.797202+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.20485v2","created_at":"2026-07-05T08:50:31.797202+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.20485","created_at":"2026-07-05T08:50:31.797202+00:00"},{"alias_kind":"pith_short_12","alias_value":"PD3MJVH5K2KH","created_at":"2026-07-05T08:50:31.797202+00:00"},{"alias_kind":"pith_short_16","alias_value":"PD3MJVH5K2KHR4JP","created_at":"2026-07-05T08:50:31.797202+00:00"},{"alias_kind":"pith_short_8","alias_value":"PD3MJVH5","created_at":"2026-07-05T08:50:31.797202+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.10098","citing_title":"Attention Sink in Transformers: A Survey on Utilization, Interpretation, and Mitigation","ref_index":153,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PD3MJVH5K2KHR4JPUGHERPLYD4","json":"https://pith.science/pith/PD3MJVH5K2KHR4JPUGHERPLYD4.json","graph_json":"https://pith.science/api/pith-number/PD3MJVH5K2KHR4JPUGHERPLYD4/graph.json","events_json":"https://pith.science/api/pith-number/PD3MJVH5K2KHR4JPUGHERPLYD4/events.json","paper":"https://pith.science/paper/PD3MJVH5"},"agent_actions":{"view_html":"https://pith.science/pith/PD3MJVH5K2KHR4JPUGHERPLYD4","download_json":"https://pith.science/pith/PD3MJVH5K2KHR4JPUGHERPLYD4.json","view_paper":"https://pith.science/paper/PD3MJVH5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.20485&json=true","fetch_graph":"https://pith.science/api/pith-number/PD3MJVH5K2KHR4JPUGHERPLYD4/graph.json","fetch_events":"https://pith.science/api/pith-number/PD3MJVH5K2KHR4JPUGHERPLYD4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PD3MJVH5K2KHR4JPUGHERPLYD4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PD3MJVH5K2KHR4JPUGHERPLYD4/action/storage_attestation","attest_author":"https://pith.science/pith/PD3MJVH5K2KHR4JPUGHERPLYD4/action/author_attestation","sign_citation":"https://pith.science/pith/PD3MJVH5K2KHR4JPUGHERPLYD4/action/citation_signature","submit_replication":"https://pith.science/pith/PD3MJVH5K2KHR4JPUGHERPLYD4/action/replication_record"}},"created_at":"2026-07-05T08:50:31.797202+00:00","updated_at":"2026-07-05T08:50:31.797202+00:00"}