{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5SGXLJW3U5CUZ4DCOSSIT4NGLZ","short_pith_number":"pith:5SGXLJW3","schema_version":"1.0","canonical_sha256":"ec8d75a6dba7454cf06274a489f1a65e71291e3c2512d45262270902f71f46cd","source":{"kind":"arxiv","id":"2409.15097","version":2},"attestation_state":"computed","paper":{"title":"Efficiently Dispatching Flash Attention For Partially Filled Attention Masks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Agniv Sharma, Jonas Geiping","submitted_at":"2024-09-23T15:11:07Z","abstract_excerpt":"Transformers are widely used across various applications, many of which yield sparse or partially filled attention matrices. Examples include attention masks designed to reduce the quadratic complexity of attention, sequence packing techniques, and recent innovations like tree masking for fast validation in MEDUSA. Despite the inherent sparsity in these matrices, the state-of-the-art algorithm Flash Attention still processes them with quadratic complexity as though they were dense. In this paper, we introduce Binary Block Masking, a highly efficient modification that enhances Flash Attention b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.15097","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-23T15:11:07Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"3310040b7d9f03fc49fc856fe8b94830d444380ae8133dcfa3411e66fbb8469a","abstract_canon_sha256":"980ae46147e935e96cda64e07b4ef2a2a6028df19d151f4f073d19077b40eeda"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:10:52.126524Z","signature_b64":"64eVyCgHbgRRzCI8iuvOxrzdv2doJexYN1/DXC5pk0in5HifQknQ7SNGtsYN2ysLDFjUb3LZcNWsDIThSo9nBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec8d75a6dba7454cf06274a489f1a65e71291e3c2512d45262270902f71f46cd","last_reissued_at":"2026-07-05T09:10:52.126038Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:10:52.126038Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficiently Dispatching Flash Attention For Partially Filled Attention Masks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Agniv Sharma, Jonas Geiping","submitted_at":"2024-09-23T15:11:07Z","abstract_excerpt":"Transformers are widely used across various applications, many of which yield sparse or partially filled attention matrices. Examples include attention masks designed to reduce the quadratic complexity of attention, sequence packing techniques, and recent innovations like tree masking for fast validation in MEDUSA. Despite the inherent sparsity in these matrices, the state-of-the-art algorithm Flash Attention still processes them with quadratic complexity as though they were dense. In this paper, we introduce Binary Block Masking, a highly efficient modification that enhances Flash Attention b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.15097","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.15097/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.15097","created_at":"2026-07-05T09:10:52.126108+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.15097v2","created_at":"2026-07-05T09:10:52.126108+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.15097","created_at":"2026-07-05T09:10:52.126108+00:00"},{"alias_kind":"pith_short_12","alias_value":"5SGXLJW3U5CU","created_at":"2026-07-05T09:10:52.126108+00:00"},{"alias_kind":"pith_short_16","alias_value":"5SGXLJW3U5CUZ4DC","created_at":"2026-07-05T09:10:52.126108+00:00"},{"alias_kind":"pith_short_8","alias_value":"5SGXLJW3","created_at":"2026-07-05T09:10:52.126108+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.01659","citing_title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5SGXLJW3U5CUZ4DCOSSIT4NGLZ","json":"https://pith.science/pith/5SGXLJW3U5CUZ4DCOSSIT4NGLZ.json","graph_json":"https://pith.science/api/pith-number/5SGXLJW3U5CUZ4DCOSSIT4NGLZ/graph.json","events_json":"https://pith.science/api/pith-number/5SGXLJW3U5CUZ4DCOSSIT4NGLZ/events.json","paper":"https://pith.science/paper/5SGXLJW3"},"agent_actions":{"view_html":"https://pith.science/pith/5SGXLJW3U5CUZ4DCOSSIT4NGLZ","download_json":"https://pith.science/pith/5SGXLJW3U5CUZ4DCOSSIT4NGLZ.json","view_paper":"https://pith.science/paper/5SGXLJW3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.15097&json=true","fetch_graph":"https://pith.science/api/pith-number/5SGXLJW3U5CUZ4DCOSSIT4NGLZ/graph.json","fetch_events":"https://pith.science/api/pith-number/5SGXLJW3U5CUZ4DCOSSIT4NGLZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5SGXLJW3U5CUZ4DCOSSIT4NGLZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5SGXLJW3U5CUZ4DCOSSIT4NGLZ/action/storage_attestation","attest_author":"https://pith.science/pith/5SGXLJW3U5CUZ4DCOSSIT4NGLZ/action/author_attestation","sign_citation":"https://pith.science/pith/5SGXLJW3U5CUZ4DCOSSIT4NGLZ/action/citation_signature","submit_replication":"https://pith.science/pith/5SGXLJW3U5CUZ4DCOSSIT4NGLZ/action/replication_record"}},"created_at":"2026-07-05T09:10:52.126108+00:00","updated_at":"2026-07-05T09:10:52.126108+00:00"}