{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YYV25GHS3RIJYSVYMUVX57YKIR","short_pith_number":"pith:YYV25GHS","schema_version":"1.0","canonical_sha256":"c62bae98f2dc509c4ab8652b7eff0a4460a272e6d760636bc9049de6cefbc2a1","source":{"kind":"arxiv","id":"2406.12016","version":2},"attestation_state":"computed","paper":{"title":"Prefixing Attention Sinks can Mitigate Activation Outliers for Large Language Model Quantization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jaeho Lee, Kyuyeun Kim, Seungwoo Son, Wonpyo Park, Woohyun Han","submitted_at":"2024-06-17T18:33:44Z","abstract_excerpt":"Despite recent advances in LLM quantization, activation quantization remains to be challenging due to the activation outliers. Conventional remedies, e.g., mixing precisions for different channels, introduce extra overhead and reduce the speedup. In this work, we develop a simple yet effective strategy to facilitate per-tensor activation quantization by preventing the generation of problematic tokens. Precisely, we propose a method to find a set of key-value cache, coined CushionCache, which mitigates outliers in subsequent tokens when inserted as a prefix. CushionCache works in two steps: Fir"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.12016","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-17T18:33:44Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"b46605b7170ea6a00241d6ddaa80d1053acbf5f4ce0304cc2f6d0efbaa470fd6","abstract_canon_sha256":"a9b27e43223b6cf04b4f02501f9fe3fd10e6c95dc53c8316c13ce4e1d350048d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:33.875759Z","signature_b64":"w3LLZ0Z8dCz5RWRfVs1SOala6VzwuS1zkeg0EkMdQszGvdhvwBNx+72wRZZFwTNyIUuBYUfKpFMEKlth1WBGBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c62bae98f2dc509c4ab8652b7eff0a4460a272e6d760636bc9049de6cefbc2a1","last_reissued_at":"2026-07-05T09:15:33.875239Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:33.875239Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Prefixing Attention Sinks can Mitigate Activation Outliers for Large Language Model Quantization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jaeho Lee, Kyuyeun Kim, Seungwoo Son, Wonpyo Park, Woohyun Han","submitted_at":"2024-06-17T18:33:44Z","abstract_excerpt":"Despite recent advances in LLM quantization, activation quantization remains to be challenging due to the activation outliers. Conventional remedies, e.g., mixing precisions for different channels, introduce extra overhead and reduce the speedup. In this work, we develop a simple yet effective strategy to facilitate per-tensor activation quantization by preventing the generation of problematic tokens. Precisely, we propose a method to find a set of key-value cache, coined CushionCache, which mitigates outliers in subsequent tokens when inserted as a prefix. CushionCache works in two steps: Fir"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.12016","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.12016/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.12016","created_at":"2026-07-05T09:15:33.875298+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.12016v2","created_at":"2026-07-05T09:15:33.875298+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12016","created_at":"2026-07-05T09:15:33.875298+00:00"},{"alias_kind":"pith_short_12","alias_value":"YYV25GHS3RIJ","created_at":"2026-07-05T09:15:33.875298+00:00"},{"alias_kind":"pith_short_16","alias_value":"YYV25GHS3RIJYSVY","created_at":"2026-07-05T09:15:33.875298+00:00"},{"alias_kind":"pith_short_8","alias_value":"YYV25GHS","created_at":"2026-07-05T09:15:33.875298+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.08504","citing_title":"A Single Layer to Explain Them All:Understanding Massive Activations in Large Language Models","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08504","citing_title":"A Single Layer to Explain Them All:Understanding Massive Activations in Large Language Models","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YYV25GHS3RIJYSVYMUVX57YKIR","json":"https://pith.science/pith/YYV25GHS3RIJYSVYMUVX57YKIR.json","graph_json":"https://pith.science/api/pith-number/YYV25GHS3RIJYSVYMUVX57YKIR/graph.json","events_json":"https://pith.science/api/pith-number/YYV25GHS3RIJYSVYMUVX57YKIR/events.json","paper":"https://pith.science/paper/YYV25GHS"},"agent_actions":{"view_html":"https://pith.science/pith/YYV25GHS3RIJYSVYMUVX57YKIR","download_json":"https://pith.science/pith/YYV25GHS3RIJYSVYMUVX57YKIR.json","view_paper":"https://pith.science/paper/YYV25GHS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.12016&json=true","fetch_graph":"https://pith.science/api/pith-number/YYV25GHS3RIJYSVYMUVX57YKIR/graph.json","fetch_events":"https://pith.science/api/pith-number/YYV25GHS3RIJYSVYMUVX57YKIR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YYV25GHS3RIJYSVYMUVX57YKIR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YYV25GHS3RIJYSVYMUVX57YKIR/action/storage_attestation","attest_author":"https://pith.science/pith/YYV25GHS3RIJYSVYMUVX57YKIR/action/author_attestation","sign_citation":"https://pith.science/pith/YYV25GHS3RIJYSVYMUVX57YKIR/action/citation_signature","submit_replication":"https://pith.science/pith/YYV25GHS3RIJYSVYMUVX57YKIR/action/replication_record"}},"created_at":"2026-07-05T09:15:33.875298+00:00","updated_at":"2026-07-05T09:15:33.875298+00:00"}