{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:XQGRNOTTGEWCM4XLR5KM5VU2KH","short_pith_number":"pith:XQGRNOTT","schema_version":"1.0","canonical_sha256":"bc0d16ba73312c2672eb8f54ced69a51c6096d3104db608518bec373353ed95a","source":{"kind":"arxiv","id":"2108.09084","version":6},"attestation_state":"computed","paper":{"title":"Fastformer: Additive Attention Can Be All You Need","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chuhan Wu, Fangzhao Wu, Tao Qi, Xing Xie, Yongfeng Huang","submitted_at":"2021-08-20T09:44:44Z","abstract_excerpt":"Transformer is a powerful model for text understanding. However, it is inefficient due to its quadratic complexity to input sequence length. Although there are many methods on Transformer acceleration, they are still either inefficient on long sequences or not effective enough. In this paper, we propose Fastformer, which is an efficient Transformer model based on additive attention. In Fastformer, instead of modeling the pair-wise interactions between tokens, we first use additive attention mechanism to model global contexts, and then further transform each token representation based on its in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.09084","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CL","submitted_at":"2021-08-20T09:44:44Z","cross_cats_sorted":[],"title_canon_sha256":"dd50c0fa2e4457a2a47b328854d3b46f91a9252bfe12a49077a79de499129e4c","abstract_canon_sha256":"af08210991cfd4f26aa66fbc8305f3c3b0a2f65d32854c38a6418e6c0d393daa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:11:32.564493Z","signature_b64":"Yg+tjiVfBWF9aMJDcidsZ6pVoTHy/0+3rleDfxiS8KKmxhMu8lOQqFgWkFu357nvEQC2ZAeD946nXCc6Oh3+Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bc0d16ba73312c2672eb8f54ced69a51c6096d3104db608518bec373353ed95a","last_reissued_at":"2026-07-05T03:11:32.564019Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:11:32.564019Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fastformer: Additive Attention Can Be All You Need","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chuhan Wu, Fangzhao Wu, Tao Qi, Xing Xie, Yongfeng Huang","submitted_at":"2021-08-20T09:44:44Z","abstract_excerpt":"Transformer is a powerful model for text understanding. However, it is inefficient due to its quadratic complexity to input sequence length. Although there are many methods on Transformer acceleration, they are still either inefficient on long sequences or not effective enough. In this paper, we propose Fastformer, which is an efficient Transformer model based on additive attention. In Fastformer, instead of modeling the pair-wise interactions between tokens, we first use additive attention mechanism to model global contexts, and then further transform each token representation based on its in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.09084","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.09084/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.09084","created_at":"2026-07-05T03:11:32.564076+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.09084v6","created_at":"2026-07-05T03:11:32.564076+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.09084","created_at":"2026-07-05T03:11:32.564076+00:00"},{"alias_kind":"pith_short_12","alias_value":"XQGRNOTTGEWC","created_at":"2026-07-05T03:11:32.564076+00:00"},{"alias_kind":"pith_short_16","alias_value":"XQGRNOTTGEWCM4XL","created_at":"2026-07-05T03:11:32.564076+00:00"},{"alias_kind":"pith_short_8","alias_value":"XQGRNOTT","created_at":"2026-07-05T03:11:32.564076+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25329","citing_title":"State Space Models Meet Remote Sensing: A Survey","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25324","citing_title":"Efficient Remote Sensing Instance Segmentation with Linear-Time State Space Distilled Visual Foundation Models","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XQGRNOTTGEWCM4XLR5KM5VU2KH","json":"https://pith.science/pith/XQGRNOTTGEWCM4XLR5KM5VU2KH.json","graph_json":"https://pith.science/api/pith-number/XQGRNOTTGEWCM4XLR5KM5VU2KH/graph.json","events_json":"https://pith.science/api/pith-number/XQGRNOTTGEWCM4XLR5KM5VU2KH/events.json","paper":"https://pith.science/paper/XQGRNOTT"},"agent_actions":{"view_html":"https://pith.science/pith/XQGRNOTTGEWCM4XLR5KM5VU2KH","download_json":"https://pith.science/pith/XQGRNOTTGEWCM4XLR5KM5VU2KH.json","view_paper":"https://pith.science/paper/XQGRNOTT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.09084&json=true","fetch_graph":"https://pith.science/api/pith-number/XQGRNOTTGEWCM4XLR5KM5VU2KH/graph.json","fetch_events":"https://pith.science/api/pith-number/XQGRNOTTGEWCM4XLR5KM5VU2KH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XQGRNOTTGEWCM4XLR5KM5VU2KH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XQGRNOTTGEWCM4XLR5KM5VU2KH/action/storage_attestation","attest_author":"https://pith.science/pith/XQGRNOTTGEWCM4XLR5KM5VU2KH/action/author_attestation","sign_citation":"https://pith.science/pith/XQGRNOTTGEWCM4XLR5KM5VU2KH/action/citation_signature","submit_replication":"https://pith.science/pith/XQGRNOTTGEWCM4XLR5KM5VU2KH/action/replication_record"}},"created_at":"2026-07-05T03:11:32.564076+00:00","updated_at":"2026-07-05T03:11:32.564076+00:00"}