{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:4EX3NNCRJPDUUP4JJI7YZCSNMY","short_pith_number":"pith:4EX3NNCR","schema_version":"1.0","canonical_sha256":"e12fb6b4514bc74a3f894a3f8c8a4d66386caad878a23ecd8b397a3741c9beff","source":{"kind":"arxiv","id":"2202.08791","version":1},"attestation_state":"computed","paper":{"title":"cosFormer: Rethinking Softmax in Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baohong Lv, Dongxu Li, Hui Deng, Junjie Yan, Lingpeng Kong, Weixuan Sun, Yiran Zhong, Yunshen Wei, Zhen Qin","submitted_at":"2022-02-17T17:53:48Z","abstract_excerpt":"Transformer has shown great successes in natural language processing, computer vision, and audio processing. As one of its core components, the softmax attention helps to capture long-range dependencies yet prohibits its scale-up due to the quadratic space and time complexity to the sequence length. Kernel methods are often adopted to reduce the complexity by approximating the softmax operator. Nevertheless, due to the approximation errors, their performances vary in different tasks/corpus and suffer crucial performance drops when compared with the vanilla softmax attention. In this paper, we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.08791","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-02-17T17:53:48Z","cross_cats_sorted":[],"title_canon_sha256":"b87589a992c5bb2f049ccab2ea1166a24a1cf45cda9409a976839cf708110f2b","abstract_canon_sha256":"69e6dda83eef2e5065d391a833ca80ed9d82d2f824d1dfb05033b3a5c1a7146c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:57:57.903699Z","signature_b64":"9mZxP0LAkLkilcw52SkvD5K/7x8f/JODa6eGed4xd0EFSSypvqqz1cErMmCRVh+ghDJ5zj9xiEAMW+gDthH1Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e12fb6b4514bc74a3f894a3f8c8a4d66386caad878a23ecd8b397a3741c9beff","last_reissued_at":"2026-07-05T03:57:57.903285Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:57:57.903285Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"cosFormer: Rethinking Softmax in Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baohong Lv, Dongxu Li, Hui Deng, Junjie Yan, Lingpeng Kong, Weixuan Sun, Yiran Zhong, Yunshen Wei, Zhen Qin","submitted_at":"2022-02-17T17:53:48Z","abstract_excerpt":"Transformer has shown great successes in natural language processing, computer vision, and audio processing. As one of its core components, the softmax attention helps to capture long-range dependencies yet prohibits its scale-up due to the quadratic space and time complexity to the sequence length. Kernel methods are often adopted to reduce the complexity by approximating the softmax operator. Nevertheless, due to the approximation errors, their performances vary in different tasks/corpus and suffer crucial performance drops when compared with the vanilla softmax attention. In this paper, we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.08791","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.08791/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.08791","created_at":"2026-07-05T03:57:57.903349+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.08791v1","created_at":"2026-07-05T03:57:57.903349+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.08791","created_at":"2026-07-05T03:57:57.903349+00:00"},{"alias_kind":"pith_short_12","alias_value":"4EX3NNCRJPDU","created_at":"2026-07-05T03:57:57.903349+00:00"},{"alias_kind":"pith_short_16","alias_value":"4EX3NNCRJPDUUP4J","created_at":"2026-07-05T03:57:57.903349+00:00"},{"alias_kind":"pith_short_8","alias_value":"4EX3NNCR","created_at":"2026-07-05T03:57:57.903349+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07706","citing_title":"The Key to Going Linear: Analysis-Driven Transformer Linearization","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2606.09862","citing_title":"Blurry Window Attention","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00746","citing_title":"Scaling Parallel Sequence Models to Foundation-Scale Vision Encoders","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04008","citing_title":"RACE Attention: A Strictly Linear-Time Attention Layer for Training on Outrageously Large Contexts","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14191","citing_title":"Attention to Mamba: A Recipe for Cross-Architecture Distillation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06473","citing_title":"MICA: Multivariate Infini Compressive Attention for Time Series Forecasting","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06473","citing_title":"MICA: Multivariate Infini Compressive Attention for Time Series Forecasting","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4EX3NNCRJPDUUP4JJI7YZCSNMY","json":"https://pith.science/pith/4EX3NNCRJPDUUP4JJI7YZCSNMY.json","graph_json":"https://pith.science/api/pith-number/4EX3NNCRJPDUUP4JJI7YZCSNMY/graph.json","events_json":"https://pith.science/api/pith-number/4EX3NNCRJPDUUP4JJI7YZCSNMY/events.json","paper":"https://pith.science/paper/4EX3NNCR"},"agent_actions":{"view_html":"https://pith.science/pith/4EX3NNCRJPDUUP4JJI7YZCSNMY","download_json":"https://pith.science/pith/4EX3NNCRJPDUUP4JJI7YZCSNMY.json","view_paper":"https://pith.science/paper/4EX3NNCR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.08791&json=true","fetch_graph":"https://pith.science/api/pith-number/4EX3NNCRJPDUUP4JJI7YZCSNMY/graph.json","fetch_events":"https://pith.science/api/pith-number/4EX3NNCRJPDUUP4JJI7YZCSNMY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4EX3NNCRJPDUUP4JJI7YZCSNMY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4EX3NNCRJPDUUP4JJI7YZCSNMY/action/storage_attestation","attest_author":"https://pith.science/pith/4EX3NNCRJPDUUP4JJI7YZCSNMY/action/author_attestation","sign_citation":"https://pith.science/pith/4EX3NNCRJPDUUP4JJI7YZCSNMY/action/citation_signature","submit_replication":"https://pith.science/pith/4EX3NNCRJPDUUP4JJI7YZCSNMY/action/replication_record"}},"created_at":"2026-07-05T03:57:57.903349+00:00","updated_at":"2026-07-05T03:57:57.903349+00:00"}