{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:QT3A2VVT5GHU574DL5RIGTRTYS","short_pith_number":"pith:QT3A2VVT","schema_version":"1.0","canonical_sha256":"84f60d56b3e98f4eff835f62834e33c4b0ffeaed574a2a769a983490af2d5600","source":{"kind":"arxiv","id":"2111.10770","version":1},"attestation_state":"computed","paper":{"title":"Efficient Softmax Approximation for Deep Neural Networks with Attention Mechanism","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ihor Vasyltsov, Wooseok Chang","submitted_at":"2021-11-21T08:56:29Z","abstract_excerpt":"There has been a rapid advance of custom hardware (HW) for accelerating the inference speed of deep neural networks (DNNs). Previously, the softmax layer was not a main concern of DNN accelerating HW, because its portion is relatively small in multi-layer perceptron or convolutional neural networks. However, as the attention mechanisms are widely used in various modern DNNs, a cost-efficient implementation of softmax layer is becoming very important. In this paper, we propose two methods to approximate softmax computation, which are based on the usage of LookUp Tables (LUTs). The required size"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.10770","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-11-21T08:56:29Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5928fc9e7a9724766f64c860f15a8b11278b7d1aee78cac0660bde547259a548","abstract_canon_sha256":"eda6da41a5ff1ce706a59b499d9056b1babb848bc8bed4203bf8b1366edc564a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:33:56.638457Z","signature_b64":"m0na7QpkM1RJYlabQe5QA3uNIv9yv7chDVL91cSws3BpDP2PL8nfAhfIE1WIe5bnkkv0XEobDBic/kViqvdtBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"84f60d56b3e98f4eff835f62834e33c4b0ffeaed574a2a769a983490af2d5600","last_reissued_at":"2026-07-05T03:33:56.637989Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:33:56.637989Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Softmax Approximation for Deep Neural Networks with Attention Mechanism","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ihor Vasyltsov, Wooseok Chang","submitted_at":"2021-11-21T08:56:29Z","abstract_excerpt":"There has been a rapid advance of custom hardware (HW) for accelerating the inference speed of deep neural networks (DNNs). Previously, the softmax layer was not a main concern of DNN accelerating HW, because its portion is relatively small in multi-layer perceptron or convolutional neural networks. However, as the attention mechanisms are widely used in various modern DNNs, a cost-efficient implementation of softmax layer is becoming very important. In this paper, we propose two methods to approximate softmax computation, which are based on the usage of LookUp Tables (LUTs). The required size"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.10770","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.10770/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.10770","created_at":"2026-07-05T03:33:56.638039+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.10770v1","created_at":"2026-07-05T03:33:56.638039+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.10770","created_at":"2026-07-05T03:33:56.638039+00:00"},{"alias_kind":"pith_short_12","alias_value":"QT3A2VVT5GHU","created_at":"2026-07-05T03:33:56.638039+00:00"},{"alias_kind":"pith_short_16","alias_value":"QT3A2VVT5GHU574D","created_at":"2026-07-05T03:33:56.638039+00:00"},{"alias_kind":"pith_short_8","alias_value":"QT3A2VVT","created_at":"2026-07-05T03:33:56.638039+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19528","citing_title":"Techniques for Peak Memory Reduction for LoRA Fine-tuning of LLMs on Edge Devices","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2407.11041","citing_title":"Integer-only Quantized Transformers for Embedded FPGA-based Time-series Forecasting in AIoT","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15944","citing_title":"CIMple: Standard-cell SRAM-based CIM with LUT-based split softmax for attention acceleration","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QT3A2VVT5GHU574DL5RIGTRTYS","json":"https://pith.science/pith/QT3A2VVT5GHU574DL5RIGTRTYS.json","graph_json":"https://pith.science/api/pith-number/QT3A2VVT5GHU574DL5RIGTRTYS/graph.json","events_json":"https://pith.science/api/pith-number/QT3A2VVT5GHU574DL5RIGTRTYS/events.json","paper":"https://pith.science/paper/QT3A2VVT"},"agent_actions":{"view_html":"https://pith.science/pith/QT3A2VVT5GHU574DL5RIGTRTYS","download_json":"https://pith.science/pith/QT3A2VVT5GHU574DL5RIGTRTYS.json","view_paper":"https://pith.science/paper/QT3A2VVT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.10770&json=true","fetch_graph":"https://pith.science/api/pith-number/QT3A2VVT5GHU574DL5RIGTRTYS/graph.json","fetch_events":"https://pith.science/api/pith-number/QT3A2VVT5GHU574DL5RIGTRTYS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QT3A2VVT5GHU574DL5RIGTRTYS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QT3A2VVT5GHU574DL5RIGTRTYS/action/storage_attestation","attest_author":"https://pith.science/pith/QT3A2VVT5GHU574DL5RIGTRTYS/action/author_attestation","sign_citation":"https://pith.science/pith/QT3A2VVT5GHU574DL5RIGTRTYS/action/citation_signature","submit_replication":"https://pith.science/pith/QT3A2VVT5GHU574DL5RIGTRTYS/action/replication_record"}},"created_at":"2026-07-05T03:33:56.638039+00:00","updated_at":"2026-07-05T03:33:56.638039+00:00"}