{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OMW77IM7LSWGR2MAG62YA4TBUY","short_pith_number":"pith:OMW77IM7","schema_version":"1.0","canonical_sha256":"732dffa19f5cac68e98037b5807261a601ae2a7b5e765c48c66247a00356346b","source":{"kind":"arxiv","id":"2411.07635","version":5},"attestation_state":"computed","paper":{"title":"Breaking the Low-Rank Dilemma of Linear Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Huaibo Huang, Qihang Fan, Ran He","submitted_at":"2024-11-12T08:30:59Z","abstract_excerpt":"The Softmax attention mechanism in Transformer models is notoriously computationally expensive, particularly due to its quadratic complexity, posing significant challenges in vision applications. In contrast, linear attention provides a far more efficient solution by reducing the complexity to linear levels. However, compared to Softmax attention, linear attention often experiences significant performance degradation. Our experiments indicate that this performance drop is due to the low-rank nature of linear attention's feature map, which hinders its ability to adequately model complex spatial"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.07635","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-12T08:30:59Z","cross_cats_sorted":[],"title_canon_sha256":"0af27f784e69ed7ff500032f5d2dfe5950f0728f39cee176576556b2a69b1328","abstract_canon_sha256":"c5cd79817d203c566fa157be0056f1ffb9c44941136bf7aa568df0856450221a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:28:26.183046Z","signature_b64":"3/9D4S/zmObKpdt778FV2cAHaFK0ZBEl/FJExjvSsG+tXdHE5YBzFD8Cg678eTLQAp/6ebUpI9vawjwictrOCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"732dffa19f5cac68e98037b5807261a601ae2a7b5e765c48c66247a00356346b","last_reissued_at":"2026-07-05T10:28:26.181349Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:28:26.181349Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Breaking the Low-Rank Dilemma of Linear Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Huaibo Huang, Qihang Fan, Ran He","submitted_at":"2024-11-12T08:30:59Z","abstract_excerpt":"The Softmax attention mechanism in Transformer models is notoriously computationally expensive, particularly due to its quadratic complexity, posing significant challenges in vision applications. In contrast, linear attention provides a far more efficient solution by reducing the complexity to linear levels. However, compared to Softmax attention, linear attention often experiences significant performance degradation. Our experiments indicate that this performance drop is due to the low-rank nature of linear attention's feature map, which hinders its ability to adequately model complex spatial"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.07635","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.07635/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.07635","created_at":"2026-07-05T10:28:26.181411+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.07635v5","created_at":"2026-07-05T10:28:26.181411+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.07635","created_at":"2026-07-05T10:28:26.181411+00:00"},{"alias_kind":"pith_short_12","alias_value":"OMW77IM7LSWG","created_at":"2026-07-05T10:28:26.181411+00:00"},{"alias_kind":"pith_short_16","alias_value":"OMW77IM7LSWGR2MA","created_at":"2026-07-05T10:28:26.181411+00:00"},{"alias_kind":"pith_short_8","alias_value":"OMW77IM7","created_at":"2026-07-05T10:28:26.181411+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.20836","citing_title":"HAD: Hybrid Architecture Distillation Outperforms Teacher in Genomic Sequence Modeling","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OMW77IM7LSWGR2MAG62YA4TBUY","json":"https://pith.science/pith/OMW77IM7LSWGR2MAG62YA4TBUY.json","graph_json":"https://pith.science/api/pith-number/OMW77IM7LSWGR2MAG62YA4TBUY/graph.json","events_json":"https://pith.science/api/pith-number/OMW77IM7LSWGR2MAG62YA4TBUY/events.json","paper":"https://pith.science/paper/OMW77IM7"},"agent_actions":{"view_html":"https://pith.science/pith/OMW77IM7LSWGR2MAG62YA4TBUY","download_json":"https://pith.science/pith/OMW77IM7LSWGR2MAG62YA4TBUY.json","view_paper":"https://pith.science/paper/OMW77IM7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.07635&json=true","fetch_graph":"https://pith.science/api/pith-number/OMW77IM7LSWGR2MAG62YA4TBUY/graph.json","fetch_events":"https://pith.science/api/pith-number/OMW77IM7LSWGR2MAG62YA4TBUY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OMW77IM7LSWGR2MAG62YA4TBUY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OMW77IM7LSWGR2MAG62YA4TBUY/action/storage_attestation","attest_author":"https://pith.science/pith/OMW77IM7LSWGR2MAG62YA4TBUY/action/author_attestation","sign_citation":"https://pith.science/pith/OMW77IM7LSWGR2MAG62YA4TBUY/action/citation_signature","submit_replication":"https://pith.science/pith/OMW77IM7LSWGR2MAG62YA4TBUY/action/replication_record"}},"created_at":"2026-07-05T10:28:26.181411+00:00","updated_at":"2026-07-05T10:28:26.181411+00:00"}