{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:J524Q4ASJXTW6VZGS6FMNLS6KX","short_pith_number":"pith:J524Q4AS","schema_version":"1.0","canonical_sha256":"4f75c870124de76f5726978ac6ae5e55ca3af772cc30d3054c47b4d7f464269e","source":{"kind":"arxiv","id":"2409.16997","version":2},"attestation_state":"computed","paper":{"title":"INT-FlashAttention: Enabling Flash Attention for INT8 Quantization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ce Zheng, Lei Su, Peizhuang Cong, Shimao Chen, Tong Yang, Yuhan Wu, Zhiying Wu, Zihan Jiang, Zirui Liu","submitted_at":"2024-09-25T15:02:25Z","abstract_excerpt":"As the foundation of large language models (LLMs), self-attention module faces the challenge of quadratic time and memory complexity with respect to sequence length. FlashAttention accelerates attention computation and reduces its memory usage by leveraging the GPU memory hierarchy. A promising research direction is to integrate FlashAttention with quantization methods. This paper introduces INT-FlashAttention, the first INT8 quantization architecture compatible with the forward workflow of FlashAttention, which significantly improves the inference speed of FlashAttention on Ampere GPUs. We im"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.16997","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-25T15:02:25Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0721b7ad69e4433fafb553425e3426c5bf7afbce5cea833dc9eb5a8188db1640","abstract_canon_sha256":"743ce8b70a745d8584d0bb575792516586a15bebed853897a8fdf81bff982e8a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:12:10.658740Z","signature_b64":"UIHh8OuDzhiCAHZLn8QZNrbecoHecWEgFzZjxqeb0rjndoj4QSLLeDZXGeLwX4nNuMgNjkJ491GEWmAF0+GkBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4f75c870124de76f5726978ac6ae5e55ca3af772cc30d3054c47b4d7f464269e","last_reissued_at":"2026-07-05T09:12:10.658180Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:12:10.658180Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"INT-FlashAttention: Enabling Flash Attention for INT8 Quantization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ce Zheng, Lei Su, Peizhuang Cong, Shimao Chen, Tong Yang, Yuhan Wu, Zhiying Wu, Zihan Jiang, Zirui Liu","submitted_at":"2024-09-25T15:02:25Z","abstract_excerpt":"As the foundation of large language models (LLMs), self-attention module faces the challenge of quadratic time and memory complexity with respect to sequence length. FlashAttention accelerates attention computation and reduces its memory usage by leveraging the GPU memory hierarchy. A promising research direction is to integrate FlashAttention with quantization methods. This paper introduces INT-FlashAttention, the first INT8 quantization architecture compatible with the forward workflow of FlashAttention, which significantly improves the inference speed of FlashAttention on Ampere GPUs. We im"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.16997","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.16997/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.16997","created_at":"2026-07-05T09:12:10.658248+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.16997v2","created_at":"2026-07-05T09:12:10.658248+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.16997","created_at":"2026-07-05T09:12:10.658248+00:00"},{"alias_kind":"pith_short_12","alias_value":"J524Q4ASJXTW","created_at":"2026-07-05T09:12:10.658248+00:00"},{"alias_kind":"pith_short_16","alias_value":"J524Q4ASJXTW6VZG","created_at":"2026-07-05T09:12:10.658248+00:00"},{"alias_kind":"pith_short_8","alias_value":"J524Q4AS","created_at":"2026-07-05T09:12:10.658248+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.25306","citing_title":"QFlash: Bridging Quantization and Memory Efficiency in Vision Transformer Attention","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J524Q4ASJXTW6VZGS6FMNLS6KX","json":"https://pith.science/pith/J524Q4ASJXTW6VZGS6FMNLS6KX.json","graph_json":"https://pith.science/api/pith-number/J524Q4ASJXTW6VZGS6FMNLS6KX/graph.json","events_json":"https://pith.science/api/pith-number/J524Q4ASJXTW6VZGS6FMNLS6KX/events.json","paper":"https://pith.science/paper/J524Q4AS"},"agent_actions":{"view_html":"https://pith.science/pith/J524Q4ASJXTW6VZGS6FMNLS6KX","download_json":"https://pith.science/pith/J524Q4ASJXTW6VZGS6FMNLS6KX.json","view_paper":"https://pith.science/paper/J524Q4AS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.16997&json=true","fetch_graph":"https://pith.science/api/pith-number/J524Q4ASJXTW6VZGS6FMNLS6KX/graph.json","fetch_events":"https://pith.science/api/pith-number/J524Q4ASJXTW6VZGS6FMNLS6KX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J524Q4ASJXTW6VZGS6FMNLS6KX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J524Q4ASJXTW6VZGS6FMNLS6KX/action/storage_attestation","attest_author":"https://pith.science/pith/J524Q4ASJXTW6VZGS6FMNLS6KX/action/author_attestation","sign_citation":"https://pith.science/pith/J524Q4ASJXTW6VZGS6FMNLS6KX/action/citation_signature","submit_replication":"https://pith.science/pith/J524Q4ASJXTW6VZGS6FMNLS6KX/action/replication_record"}},"created_at":"2026-07-05T09:12:10.658248+00:00","updated_at":"2026-07-05T09:12:10.658248+00:00"}