{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3FYM47HOHUA7M66GGKINWLDZRA","short_pith_number":"pith:3FYM47HO","schema_version":"1.0","canonical_sha256":"d970ce7cee3d01f67bc63290db2c79883f8c4745a56b77a5485893ab1487f54d","source":{"kind":"arxiv","id":"2504.20938","version":1},"attestation_state":"computed","paper":{"title":"Towards Understanding the Nature of Attention with Low-Rank Sparse Decomposition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Junping Zhang, Junxuan Wang, Qiong Tang, Rui Lin, Wentao Shu, Xipeng Qiu, Xuyang Ge, Zhengfu He","submitted_at":"2025-04-29T17:03:03Z","abstract_excerpt":"We propose Low-Rank Sparse Attention (Lorsa), a sparse replacement model of Transformer attention layers to disentangle original Multi Head Self Attention (MHSA) into individually comprehensible components. Lorsa is designed to address the challenge of attention superposition to understand attention-mediated interaction between features in different token positions. We show that Lorsa heads find cleaner and finer-grained versions of previously discovered MHSA behaviors like induction heads, successor heads and attention sink behavior (i.e., heavily attending to the first token). Lorsa and Spar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.20938","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-29T17:03:03Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"3f77ab5d19baca226ef2a5e3046f53b2ef0655475e7f0e23bdb1473bdb18a6c3","abstract_canon_sha256":"cfc8089198da9dc8768969575059c68edd315424aaf6607416ab872684f3d772"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:55:51.856666Z","signature_b64":"OAaifWd3L+oFTZYpgDEGR0i3HX1pyP5Vpr8L2ixQDNJebFph9Dy4K7Z39a2/75B8CAi8VWy2T3r9wEIbp8DNDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d970ce7cee3d01f67bc63290db2c79883f8c4745a56b77a5485893ab1487f54d","last_reissued_at":"2026-07-05T10:55:51.856172Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:55:51.856172Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Understanding the Nature of Attention with Low-Rank Sparse Decomposition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Junping Zhang, Junxuan Wang, Qiong Tang, Rui Lin, Wentao Shu, Xipeng Qiu, Xuyang Ge, Zhengfu He","submitted_at":"2025-04-29T17:03:03Z","abstract_excerpt":"We propose Low-Rank Sparse Attention (Lorsa), a sparse replacement model of Transformer attention layers to disentangle original Multi Head Self Attention (MHSA) into individually comprehensible components. Lorsa is designed to address the challenge of attention superposition to understand attention-mediated interaction between features in different token positions. We show that Lorsa heads find cleaner and finer-grained versions of previously discovered MHSA behaviors like induction heads, successor heads and attention sink behavior (i.e., heavily attending to the first token). Lorsa and Spar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.20938","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.20938/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.20938","created_at":"2026-07-05T10:55:51.856227+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.20938v1","created_at":"2026-07-05T10:55:51.856227+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.20938","created_at":"2026-07-05T10:55:51.856227+00:00"},{"alias_kind":"pith_short_12","alias_value":"3FYM47HOHUA7","created_at":"2026-07-05T10:55:51.856227+00:00"},{"alias_kind":"pith_short_16","alias_value":"3FYM47HOHUA7M66G","created_at":"2026-07-05T10:55:51.856227+00:00"},{"alias_kind":"pith_short_8","alias_value":"3FYM47HO","created_at":"2026-07-05T10:55:51.856227+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07604","citing_title":"Contribution Weights: A Geometrical Analysis of Self-Attention Transformers","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12770","citing_title":"WriteSAE: Sparse Autoencoders for Recurrent State","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12770","citing_title":"WriteSAE: Sparse Autoencoders for Recurrent State","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11297","citing_title":"The Past Is Not Past: Memory-Enhanced Dynamic Reward Shaping","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3FYM47HOHUA7M66GGKINWLDZRA","json":"https://pith.science/pith/3FYM47HOHUA7M66GGKINWLDZRA.json","graph_json":"https://pith.science/api/pith-number/3FYM47HOHUA7M66GGKINWLDZRA/graph.json","events_json":"https://pith.science/api/pith-number/3FYM47HOHUA7M66GGKINWLDZRA/events.json","paper":"https://pith.science/paper/3FYM47HO"},"agent_actions":{"view_html":"https://pith.science/pith/3FYM47HOHUA7M66GGKINWLDZRA","download_json":"https://pith.science/pith/3FYM47HOHUA7M66GGKINWLDZRA.json","view_paper":"https://pith.science/paper/3FYM47HO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.20938&json=true","fetch_graph":"https://pith.science/api/pith-number/3FYM47HOHUA7M66GGKINWLDZRA/graph.json","fetch_events":"https://pith.science/api/pith-number/3FYM47HOHUA7M66GGKINWLDZRA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3FYM47HOHUA7M66GGKINWLDZRA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3FYM47HOHUA7M66GGKINWLDZRA/action/storage_attestation","attest_author":"https://pith.science/pith/3FYM47HOHUA7M66GGKINWLDZRA/action/author_attestation","sign_citation":"https://pith.science/pith/3FYM47HOHUA7M66GGKINWLDZRA/action/citation_signature","submit_replication":"https://pith.science/pith/3FYM47HOHUA7M66GGKINWLDZRA/action/replication_record"}},"created_at":"2026-07-05T10:55:51.856227+00:00","updated_at":"2026-07-05T10:55:51.856227+00:00"}