{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6MJZWBG3TPKNQPIZMO43CIV7UY","short_pith_number":"pith:6MJZWBG3","schema_version":"1.0","canonical_sha256":"f3139b04db9bd4d83d1963b9b122bfa601134c53f991aa039c788700a48fd215","source":{"kind":"arxiv","id":"2410.10254","version":3},"attestation_state":"computed","paper":{"title":"LoLCATs: On Low-Rank Linearizing of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aaryan Singhal, Alan Wu, Benjamin Spector, Christopher R\\'e, Krithik Ramesh, Michael Zhang, Rahul Chalamala, Simran Arora","submitted_at":"2024-10-14T08:10:34Z","abstract_excerpt":"Recent works show we can linearize large language models (LLMs) -- swapping the quadratic attentions of popular Transformer-based LLMs with subquadratic analogs, such as linear attention -- avoiding the expensive pretraining costs. However, linearizing LLMs often significantly degrades model quality, still requires training over billions of tokens, and remains limited to smaller 1.3B to 7B LLMs. We thus propose Low-rank Linear Conversion via Attention Transfer (LoLCATs), a simple two-step method that improves LLM linearizing quality with orders of magnitudes less memory and compute. We base th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.10254","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-14T08:10:34Z","cross_cats_sorted":["cs.AI","cs.CL","stat.ML"],"title_canon_sha256":"1429d0a450a16eea11cdb0d596c2b66392b99eb570263bf2f89514cd5d54fe37","abstract_canon_sha256":"75cafaadf3adaaaa306139a842bd32ceffade60cc47e668f0b4c7baffde8ca1d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:25:11.673204Z","signature_b64":"tAFfcQEL/sYSxaV7yimg1+AEyjkdzgwu4/YLT8yCmihDL9ysP1wEo0qwG2wvUVQ2Os4ZGsbmC7cf8HZo7de+Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f3139b04db9bd4d83d1963b9b122bfa601134c53f991aa039c788700a48fd215","last_reissued_at":"2026-07-05T10:25:11.672126Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:25:11.672126Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LoLCATs: On Low-Rank Linearizing of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aaryan Singhal, Alan Wu, Benjamin Spector, Christopher R\\'e, Krithik Ramesh, Michael Zhang, Rahul Chalamala, Simran Arora","submitted_at":"2024-10-14T08:10:34Z","abstract_excerpt":"Recent works show we can linearize large language models (LLMs) -- swapping the quadratic attentions of popular Transformer-based LLMs with subquadratic analogs, such as linear attention -- avoiding the expensive pretraining costs. However, linearizing LLMs often significantly degrades model quality, still requires training over billions of tokens, and remains limited to smaller 1.3B to 7B LLMs. We thus propose Low-rank Linear Conversion via Attention Transfer (LoLCATs), a simple two-step method that improves LLM linearizing quality with orders of magnitudes less memory and compute. We base th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.10254","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.10254/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.10254","created_at":"2026-07-05T10:25:11.672295+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.10254v3","created_at":"2026-07-05T10:25:11.672295+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.10254","created_at":"2026-07-05T10:25:11.672295+00:00"},{"alias_kind":"pith_short_12","alias_value":"6MJZWBG3TPKN","created_at":"2026-07-05T10:25:11.672295+00:00"},{"alias_kind":"pith_short_16","alias_value":"6MJZWBG3TPKNQPIZ","created_at":"2026-07-05T10:25:11.672295+00:00"},{"alias_kind":"pith_short_8","alias_value":"6MJZWBG3","created_at":"2026-07-05T10:25:11.672295+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07706","citing_title":"The Key to Going Linear: Analysis-Driven Transformer Linearization","ref_index":26,"is_internal_anchor":true},{"citing_arxiv_id":"2606.30562","citing_title":"Morphing into Hybrid Attention Models","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2410.13846","citing_title":"LightTransfer: Your Long-Context LLM is Secretly a Hybrid Model with Effortless Adaptation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04800","citing_title":"Hybrid Architectures for Language Models: Systematic Analysis and Design Insights","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04595","citing_title":"SpikingMamba: Towards Energy-Efficient Large Language Models via Knowledge Distillation from Mamba","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13189","citing_title":"MoBA: Mixture of Block Attention for Long-Context LLMs","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26692","citing_title":"Kimi Linear: An Expressive, Efficient Attention Architecture","ref_index":126,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14191","citing_title":"Attention to Mamba: A Recipe for Cross-Architecture Distillation","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06523","citing_title":"On the Implicit Reward Overfitting and the Low-rank Dynamics in RLVR","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02772","citing_title":"Linearizing Vision Transformer with Test-Time Training","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6MJZWBG3TPKNQPIZMO43CIV7UY","json":"https://pith.science/pith/6MJZWBG3TPKNQPIZMO43CIV7UY.json","graph_json":"https://pith.science/api/pith-number/6MJZWBG3TPKNQPIZMO43CIV7UY/graph.json","events_json":"https://pith.science/api/pith-number/6MJZWBG3TPKNQPIZMO43CIV7UY/events.json","paper":"https://pith.science/paper/6MJZWBG3"},"agent_actions":{"view_html":"https://pith.science/pith/6MJZWBG3TPKNQPIZMO43CIV7UY","download_json":"https://pith.science/pith/6MJZWBG3TPKNQPIZMO43CIV7UY.json","view_paper":"https://pith.science/paper/6MJZWBG3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.10254&json=true","fetch_graph":"https://pith.science/api/pith-number/6MJZWBG3TPKNQPIZMO43CIV7UY/graph.json","fetch_events":"https://pith.science/api/pith-number/6MJZWBG3TPKNQPIZMO43CIV7UY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6MJZWBG3TPKNQPIZMO43CIV7UY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6MJZWBG3TPKNQPIZMO43CIV7UY/action/storage_attestation","attest_author":"https://pith.science/pith/6MJZWBG3TPKNQPIZMO43CIV7UY/action/author_attestation","sign_citation":"https://pith.science/pith/6MJZWBG3TPKNQPIZMO43CIV7UY/action/citation_signature","submit_replication":"https://pith.science/pith/6MJZWBG3TPKNQPIZMO43CIV7UY/action/replication_record"}},"created_at":"2026-07-05T10:25:11.672295+00:00","updated_at":"2026-07-05T10:25:11.672295+00:00"}