{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:P6B6MKMIZJKB2VN7K3ZOG7T6W2","short_pith_number":"pith:P6B6MKMI","schema_version":"1.0","canonical_sha256":"7f83e62988ca541d55bf56f2e37e7eb6a67bbf5a0c8285e0423ebc46fd1d0b58","source":{"kind":"arxiv","id":"2505.08098","version":1},"attestation_state":"computed","paper":{"title":"Fused3S: Fast Sparse Attention on Tensor Cores","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Aparna Chandramowlishwaran, Zitong Li","submitted_at":"2025-05-12T22:09:05Z","abstract_excerpt":"Sparse attention is a core building block in many leading neural network models, from graph-structured learning to sparse sequence modeling. It can be decomposed into a sequence of three sparse matrix operations (3S): sampled dense-dense matrix multiplication (SDDMM), softmax normalization, and sparse matrix multiplication (SpMM). Efficiently executing the 3S computational pattern on modern GPUs remains challenging due to (a) the mismatch between unstructured sparsity and tensor cores optimized for dense operations, and (b) the high cost of data movement. Previous works have optimized these sp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.08098","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2025-05-12T22:09:05Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"6478bd18dd9b8ef8c155b592a904e20019be8a970ae5bfbdf01ef6e670c464ec","abstract_canon_sha256":"870aec69dc25fb460839ddd6e1c3ed299d33be7b6cdd32920b993b9058eaf1df"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:02:02.611578Z","signature_b64":"GEjji955oQfMORIVPKqVt4PEgVyecK/xSbRdF4i5Ha8xkyIeRMBh8GflZdYMm9leur2WmyHkiypwmQpfmZbrBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7f83e62988ca541d55bf56f2e37e7eb6a67bbf5a0c8285e0423ebc46fd1d0b58","last_reissued_at":"2026-07-05T11:02:02.611103Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:02:02.611103Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fused3S: Fast Sparse Attention on Tensor Cores","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Aparna Chandramowlishwaran, Zitong Li","submitted_at":"2025-05-12T22:09:05Z","abstract_excerpt":"Sparse attention is a core building block in many leading neural network models, from graph-structured learning to sparse sequence modeling. It can be decomposed into a sequence of three sparse matrix operations (3S): sampled dense-dense matrix multiplication (SDDMM), softmax normalization, and sparse matrix multiplication (SpMM). Efficiently executing the 3S computational pattern on modern GPUs remains challenging due to (a) the mismatch between unstructured sparsity and tensor cores optimized for dense operations, and (b) the high cost of data movement. Previous works have optimized these sp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.08098","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.08098/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.08098","created_at":"2026-07-05T11:02:02.611188+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.08098v1","created_at":"2026-07-05T11:02:02.611188+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.08098","created_at":"2026-07-05T11:02:02.611188+00:00"},{"alias_kind":"pith_short_12","alias_value":"P6B6MKMIZJKB","created_at":"2026-07-05T11:02:02.611188+00:00"},{"alias_kind":"pith_short_16","alias_value":"P6B6MKMIZJKB2VN7","created_at":"2026-07-05T11:02:02.611188+00:00"},{"alias_kind":"pith_short_8","alias_value":"P6B6MKMI","created_at":"2026-07-05T11:02:02.611188+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31500","citing_title":"On Efficient Scaling of GNNs via IO-Aware Layers Implementations","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P6B6MKMIZJKB2VN7K3ZOG7T6W2","json":"https://pith.science/pith/P6B6MKMIZJKB2VN7K3ZOG7T6W2.json","graph_json":"https://pith.science/api/pith-number/P6B6MKMIZJKB2VN7K3ZOG7T6W2/graph.json","events_json":"https://pith.science/api/pith-number/P6B6MKMIZJKB2VN7K3ZOG7T6W2/events.json","paper":"https://pith.science/paper/P6B6MKMI"},"agent_actions":{"view_html":"https://pith.science/pith/P6B6MKMIZJKB2VN7K3ZOG7T6W2","download_json":"https://pith.science/pith/P6B6MKMIZJKB2VN7K3ZOG7T6W2.json","view_paper":"https://pith.science/paper/P6B6MKMI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.08098&json=true","fetch_graph":"https://pith.science/api/pith-number/P6B6MKMIZJKB2VN7K3ZOG7T6W2/graph.json","fetch_events":"https://pith.science/api/pith-number/P6B6MKMIZJKB2VN7K3ZOG7T6W2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P6B6MKMIZJKB2VN7K3ZOG7T6W2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P6B6MKMIZJKB2VN7K3ZOG7T6W2/action/storage_attestation","attest_author":"https://pith.science/pith/P6B6MKMIZJKB2VN7K3ZOG7T6W2/action/author_attestation","sign_citation":"https://pith.science/pith/P6B6MKMIZJKB2VN7K3ZOG7T6W2/action/citation_signature","submit_replication":"https://pith.science/pith/P6B6MKMIZJKB2VN7K3ZOG7T6W2/action/replication_record"}},"created_at":"2026-07-05T11:02:02.611188+00:00","updated_at":"2026-07-05T11:02:02.611188+00:00"}