{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:R5OG23V6PLZEKDD33HWB75DI3W","short_pith_number":"pith:R5OG23V6","schema_version":"1.0","canonical_sha256":"8f5c6d6ebe7af2450c7bd9ec1ff468dd9452f577cdece5bad53537c1b9f151bf","source":{"kind":"arxiv","id":"2109.12188","version":2},"attestation_state":"computed","paper":{"title":"Predicting Attention Sparsity in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andr\\'e F. T. Martins, Ant\\'onio G\\'ois, Erick Fonseca, Marcos Treviso, Patrick Fernandes","submitted_at":"2021-09-24T20:51:21Z","abstract_excerpt":"Transformers' quadratic complexity with respect to the input sequence length has motivated a body of work on efficient sparse approximations to softmax. An alternative path, used by entmax transformers, consists of having built-in exact sparse attention; however this approach still requires quadratic computation. In this paper, we propose Sparsefinder, a simple model trained to identify the sparsity pattern of entmax attention before computing it. We experiment with three variants of our method, based on distances, quantization, and clustering, on two tasks: machine translation (attention in t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.12188","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-09-24T20:51:21Z","cross_cats_sorted":[],"title_canon_sha256":"6ed149b7feaa6a1cb547c282512c8d797a8ce376a0d70e33aef71fa2c6af954f","abstract_canon_sha256":"732c0971db9cf58c55c6d4e0e260cc51eb161a2616598ab65f5492e2ae1d2e13"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:16:36.706294Z","signature_b64":"QttjTzxnvc52Mot/GqCI5FBfax6czZ5IDS++6B3ViMC55l6VBLDQHdpg3w53rRyxY2Z+Lb1xr5V0tF/3g7C3AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8f5c6d6ebe7af2450c7bd9ec1ff468dd9452f577cdece5bad53537c1b9f151bf","last_reissued_at":"2026-07-05T04:16:36.705852Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:16:36.705852Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Predicting Attention Sparsity in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andr\\'e F. T. Martins, Ant\\'onio G\\'ois, Erick Fonseca, Marcos Treviso, Patrick Fernandes","submitted_at":"2021-09-24T20:51:21Z","abstract_excerpt":"Transformers' quadratic complexity with respect to the input sequence length has motivated a body of work on efficient sparse approximations to softmax. An alternative path, used by entmax transformers, consists of having built-in exact sparse attention; however this approach still requires quadratic computation. In this paper, we propose Sparsefinder, a simple model trained to identify the sparsity pattern of entmax attention before computing it. We experiment with three variants of our method, based on distances, quantization, and clustering, on two tasks: machine translation (attention in t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.12188","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.12188/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.12188","created_at":"2026-07-05T04:16:36.705916+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.12188v2","created_at":"2026-07-05T04:16:36.705916+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.12188","created_at":"2026-07-05T04:16:36.705916+00:00"},{"alias_kind":"pith_short_12","alias_value":"R5OG23V6PLZE","created_at":"2026-07-05T04:16:36.705916+00:00"},{"alias_kind":"pith_short_16","alias_value":"R5OG23V6PLZEKDD3","created_at":"2026-07-05T04:16:36.705916+00:00"},{"alias_kind":"pith_short_8","alias_value":"R5OG23V6","created_at":"2026-07-05T04:16:36.705916+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.01223","citing_title":"Perspectives on Tsallis Statistics for Artificial Intelligence","ref_index":35,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R5OG23V6PLZEKDD33HWB75DI3W","json":"https://pith.science/pith/R5OG23V6PLZEKDD33HWB75DI3W.json","graph_json":"https://pith.science/api/pith-number/R5OG23V6PLZEKDD33HWB75DI3W/graph.json","events_json":"https://pith.science/api/pith-number/R5OG23V6PLZEKDD33HWB75DI3W/events.json","paper":"https://pith.science/paper/R5OG23V6"},"agent_actions":{"view_html":"https://pith.science/pith/R5OG23V6PLZEKDD33HWB75DI3W","download_json":"https://pith.science/pith/R5OG23V6PLZEKDD33HWB75DI3W.json","view_paper":"https://pith.science/paper/R5OG23V6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.12188&json=true","fetch_graph":"https://pith.science/api/pith-number/R5OG23V6PLZEKDD33HWB75DI3W/graph.json","fetch_events":"https://pith.science/api/pith-number/R5OG23V6PLZEKDD33HWB75DI3W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R5OG23V6PLZEKDD33HWB75DI3W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R5OG23V6PLZEKDD33HWB75DI3W/action/storage_attestation","attest_author":"https://pith.science/pith/R5OG23V6PLZEKDD33HWB75DI3W/action/author_attestation","sign_citation":"https://pith.science/pith/R5OG23V6PLZEKDD33HWB75DI3W/action/citation_signature","submit_replication":"https://pith.science/pith/R5OG23V6PLZEKDD33HWB75DI3W/action/replication_record"}},"created_at":"2026-07-05T04:16:36.705916+00:00","updated_at":"2026-07-05T04:16:36.705916+00:00"}