{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:R3BCBVYCL6WG2VVENDGCXFLYGO","short_pith_number":"pith:R3BCBVYC","schema_version":"1.0","canonical_sha256":"8ec220d7025fac6d56a468cc2b9578339223b5b23c81a1889b8bc3c3879daaa8","source":{"kind":"arxiv","id":"2505.14201","version":1},"attestation_state":"computed","paper":{"title":"FLASH-D: FlashAttention with Hidden Softmax Division","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.AR"],"primary_cat":"cs.LG","authors_text":"Giorgos Dimitrakopoulos, Kosmas Alexandridis, Vasileios Titopoulos","submitted_at":"2025-05-20T11:01:33Z","abstract_excerpt":"The transformer's attention mechanism has revolutionized AI and machine learning, with its efficient computation being crucial to its performance. However, calculating attention involves matrix operations interspersed with softmax rescaling, which inherently slows down computation and requires processing the entire input sequence. Building on online softmax computation, FlashAttention integrates softmax calculation with matrix arithmetic, enabling tiled computation independent of sequence length. While optimized for GPUs, FlashAttention's simplicity makes it amenable to direct hardware acceler"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.14201","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-20T11:01:33Z","cross_cats_sorted":["cs.AI","cs.AR"],"title_canon_sha256":"1963563b9a41f98153f448062cdb316177926a4cf679edf8b39473d08d80aefa","abstract_canon_sha256":"5176ffca6c54a2507c82e88eeb88b35e087986af6b5a97be520d82e9fe392148"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:05:57.712461Z","signature_b64":"Au+N2xyRv7imH2UMDh0bRQ26teSU5Rn2PkSo6Z8RybVOWcHeCpTys5kAdXVjVQstxLQm1lhNPF23ikAKZI6LBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8ec220d7025fac6d56a468cc2b9578339223b5b23c81a1889b8bc3c3879daaa8","last_reissued_at":"2026-07-05T11:05:57.711958Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:05:57.711958Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FLASH-D: FlashAttention with Hidden Softmax Division","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.AR"],"primary_cat":"cs.LG","authors_text":"Giorgos Dimitrakopoulos, Kosmas Alexandridis, Vasileios Titopoulos","submitted_at":"2025-05-20T11:01:33Z","abstract_excerpt":"The transformer's attention mechanism has revolutionized AI and machine learning, with its efficient computation being crucial to its performance. However, calculating attention involves matrix operations interspersed with softmax rescaling, which inherently slows down computation and requires processing the entire input sequence. Building on online softmax computation, FlashAttention integrates softmax calculation with matrix arithmetic, enabling tiled computation independent of sequence length. While optimized for GPUs, FlashAttention's simplicity makes it amenable to direct hardware acceler"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.14201","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.14201/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.14201","created_at":"2026-07-05T11:05:57.712016+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.14201v1","created_at":"2026-07-05T11:05:57.712016+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.14201","created_at":"2026-07-05T11:05:57.712016+00:00"},{"alias_kind":"pith_short_12","alias_value":"R3BCBVYCL6WG","created_at":"2026-07-05T11:05:57.712016+00:00"},{"alias_kind":"pith_short_16","alias_value":"R3BCBVYCL6WG2VVE","created_at":"2026-07-05T11:05:57.712016+00:00"},{"alias_kind":"pith_short_8","alias_value":"R3BCBVYC","created_at":"2026-07-05T11:05:57.712016+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.12798","citing_title":"VFA: Relieving Vector Operations in Flash Attention with Global Maximum Pre-computation","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R3BCBVYCL6WG2VVENDGCXFLYGO","json":"https://pith.science/pith/R3BCBVYCL6WG2VVENDGCXFLYGO.json","graph_json":"https://pith.science/api/pith-number/R3BCBVYCL6WG2VVENDGCXFLYGO/graph.json","events_json":"https://pith.science/api/pith-number/R3BCBVYCL6WG2VVENDGCXFLYGO/events.json","paper":"https://pith.science/paper/R3BCBVYC"},"agent_actions":{"view_html":"https://pith.science/pith/R3BCBVYCL6WG2VVENDGCXFLYGO","download_json":"https://pith.science/pith/R3BCBVYCL6WG2VVENDGCXFLYGO.json","view_paper":"https://pith.science/paper/R3BCBVYC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.14201&json=true","fetch_graph":"https://pith.science/api/pith-number/R3BCBVYCL6WG2VVENDGCXFLYGO/graph.json","fetch_events":"https://pith.science/api/pith-number/R3BCBVYCL6WG2VVENDGCXFLYGO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R3BCBVYCL6WG2VVENDGCXFLYGO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R3BCBVYCL6WG2VVENDGCXFLYGO/action/storage_attestation","attest_author":"https://pith.science/pith/R3BCBVYCL6WG2VVENDGCXFLYGO/action/author_attestation","sign_citation":"https://pith.science/pith/R3BCBVYCL6WG2VVENDGCXFLYGO/action/citation_signature","submit_replication":"https://pith.science/pith/R3BCBVYCL6WG2VVENDGCXFLYGO/action/replication_record"}},"created_at":"2026-07-05T11:05:57.712016+00:00","updated_at":"2026-07-05T11:05:57.712016+00:00"}