{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:YPESEIV7EIQPR2C3QR6LGCKJUI","short_pith_number":"pith:YPESEIV7","schema_version":"1.0","canonical_sha256":"c3c92222bf2220f8e85b847cb30949a22b8699c19dfe98a13fec6257eab6b71a","source":{"kind":"arxiv","id":"2103.02143","version":2},"attestation_state":"computed","paper":{"title":"Random Feature Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dani Yogatama, Hao Peng, Lingpeng Kong, Nikolaos Pappas, Noah A. Smith, Roy Schwartz","submitted_at":"2021-03-03T02:48:56Z","abstract_excerpt":"Transformers are state-of-the-art models for a variety of sequence modeling tasks. At their core is an attention function which models pairwise interactions between the inputs at every timestep. While attention is powerful, it does not scale efficiently to long sequences due to its quadratic time and space complexity in the sequence length. We propose RFA, a linear time and space attention that uses random feature methods to approximate the softmax function, and explore its application in transformers. RFA can be used as a drop-in replacement for conventional softmax attention and offers a str"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2103.02143","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-03-03T02:48:56Z","cross_cats_sorted":[],"title_canon_sha256":"b68959ff6ec70c27991dfb3718d0a9e5f8ccdde9d4261c3cfb4baa76e914d910","abstract_canon_sha256":"41a7fcc773cdd726296e28d2d8fe7986a8ae96b8beb81172e3f7c45644d3d8c0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:24:54.518447Z","signature_b64":"E9A+/CUCy3QiZe1xZPbRno7Uf9WnTsa8cfYnpYNHwBPzaCVTGwNLtr96OVMemeDJACQKoJwfXT/TnsIEUthaCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c3c92222bf2220f8e85b847cb30949a22b8699c19dfe98a13fec6257eab6b71a","last_reissued_at":"2026-07-05T02:24:54.517944Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:24:54.517944Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Random Feature Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dani Yogatama, Hao Peng, Lingpeng Kong, Nikolaos Pappas, Noah A. Smith, Roy Schwartz","submitted_at":"2021-03-03T02:48:56Z","abstract_excerpt":"Transformers are state-of-the-art models for a variety of sequence modeling tasks. At their core is an attention function which models pairwise interactions between the inputs at every timestep. While attention is powerful, it does not scale efficiently to long sequences due to its quadratic time and space complexity in the sequence length. We propose RFA, a linear time and space attention that uses random feature methods to approximate the softmax function, and explore its application in transformers. RFA can be used as a drop-in replacement for conventional softmax attention and offers a str"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.02143","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.02143/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2103.02143","created_at":"2026-07-05T02:24:54.518001+00:00"},{"alias_kind":"arxiv_version","alias_value":"2103.02143v2","created_at":"2026-07-05T02:24:54.518001+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.02143","created_at":"2026-07-05T02:24:54.518001+00:00"},{"alias_kind":"pith_short_12","alias_value":"YPESEIV7EIQP","created_at":"2026-07-05T02:24:54.518001+00:00"},{"alias_kind":"pith_short_16","alias_value":"YPESEIV7EIQPR2C3","created_at":"2026-07-05T02:24:54.518001+00:00"},{"alias_kind":"pith_short_8","alias_value":"YPESEIV7","created_at":"2026-07-05T02:24:54.518001+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23159","citing_title":"General-Purpose Nonlinear Function Approximation via Linear Integrated Photonics","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09862","citing_title":"Blurry Window Attention","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28769","citing_title":"Multi-Mixer Models: Flexible Sequence Modeling with Shared Representations","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00746","citing_title":"Scaling Parallel Sequence Models to Foundation-Scale Vision Encoders","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2510.27258","citing_title":"Higher-order Linear Attention","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14360","citing_title":"M$^2$RNN: Non-Linear RNNs with Matrix-Valued States for Scalable Language Modeling","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2312.06635","citing_title":"Gated Linear Attention Transformers with Hardware-Efficient Training","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14191","citing_title":"Attention to Mamba: A Recipe for Cross-Architecture Distillation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12491","citing_title":"Elastic Attention Cores for Scalable Vision Transformers","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09905","citing_title":"Rethinking Random Transformers as Adaptive Sequence Smoothers for Sleep Staging","ref_index":81,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YPESEIV7EIQPR2C3QR6LGCKJUI","json":"https://pith.science/pith/YPESEIV7EIQPR2C3QR6LGCKJUI.json","graph_json":"https://pith.science/api/pith-number/YPESEIV7EIQPR2C3QR6LGCKJUI/graph.json","events_json":"https://pith.science/api/pith-number/YPESEIV7EIQPR2C3QR6LGCKJUI/events.json","paper":"https://pith.science/paper/YPESEIV7"},"agent_actions":{"view_html":"https://pith.science/pith/YPESEIV7EIQPR2C3QR6LGCKJUI","download_json":"https://pith.science/pith/YPESEIV7EIQPR2C3QR6LGCKJUI.json","view_paper":"https://pith.science/paper/YPESEIV7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2103.02143&json=true","fetch_graph":"https://pith.science/api/pith-number/YPESEIV7EIQPR2C3QR6LGCKJUI/graph.json","fetch_events":"https://pith.science/api/pith-number/YPESEIV7EIQPR2C3QR6LGCKJUI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YPESEIV7EIQPR2C3QR6LGCKJUI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YPESEIV7EIQPR2C3QR6LGCKJUI/action/storage_attestation","attest_author":"https://pith.science/pith/YPESEIV7EIQPR2C3QR6LGCKJUI/action/author_attestation","sign_citation":"https://pith.science/pith/YPESEIV7EIQPR2C3QR6LGCKJUI/action/citation_signature","submit_replication":"https://pith.science/pith/YPESEIV7EIQPR2C3QR6LGCKJUI/action/replication_record"}},"created_at":"2026-07-05T02:24:54.518001+00:00","updated_at":"2026-07-05T02:24:54.518001+00:00"}