{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KN4PYLFXIGK57LCX22YPNBZLC6","short_pith_number":"pith:KN4PYLFX","schema_version":"1.0","canonical_sha256":"5378fc2cb74195dfac57d6b0f6872b178d74ec1196bff9a699a6543575770238","source":{"kind":"arxiv","id":"2404.09695","version":1},"attestation_state":"computed","paper":{"title":"LoRAP: Transformer Sub-Layers Deserve Differentiated Structured Compression for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Guangyan Li, Wensheng Zhang, Yongqiang Tang","submitted_at":"2024-04-15T11:53:22Z","abstract_excerpt":"Large language models (LLMs) show excellent performance in difficult tasks, but they often require massive memories and computational resources. How to reduce the parameter scale of LLMs has become research hotspots. In this study, we make an important observation that the multi-head self-attention (MHA) sub-layer of Transformer exhibits noticeable low-rank structure, while the feed-forward network (FFN) sub-layer does not. With this regard, we design a mixed compression model, which organically combines Low-Rank matrix approximation And structured Pruning (LoRAP). For the MHA sub-layer, we pr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.09695","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-15T11:53:22Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"f09bba9b3b92a36816bae8560eb511f36ad7f5e043d9560e650f8ff487dc086b","abstract_canon_sha256":"16edfad10a7d4bb7ab6f4dcd92e952b84f019e545ee9a53c2724f7bda14178bd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:08:10.476380Z","signature_b64":"Y+LNdiNqn8jkIWd6GTsG2HfaLy9wrDbLslPmgyQ8TB0dqONVOvfpJNUvWD0Pc/+z0q6kWzJbmZerbFIfyBGLAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5378fc2cb74195dfac57d6b0f6872b178d74ec1196bff9a699a6543575770238","last_reissued_at":"2026-07-05T08:08:10.475822Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:08:10.475822Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LoRAP: Transformer Sub-Layers Deserve Differentiated Structured Compression for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Guangyan Li, Wensheng Zhang, Yongqiang Tang","submitted_at":"2024-04-15T11:53:22Z","abstract_excerpt":"Large language models (LLMs) show excellent performance in difficult tasks, but they often require massive memories and computational resources. How to reduce the parameter scale of LLMs has become research hotspots. In this study, we make an important observation that the multi-head self-attention (MHA) sub-layer of Transformer exhibits noticeable low-rank structure, while the feed-forward network (FFN) sub-layer does not. With this regard, we design a mixed compression model, which organically combines Low-Rank matrix approximation And structured Pruning (LoRAP). For the MHA sub-layer, we pr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.09695","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.09695/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.09695","created_at":"2026-07-05T08:08:10.475884+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.09695v1","created_at":"2026-07-05T08:08:10.475884+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.09695","created_at":"2026-07-05T08:08:10.475884+00:00"},{"alias_kind":"pith_short_12","alias_value":"KN4PYLFXIGK5","created_at":"2026-07-05T08:08:10.475884+00:00"},{"alias_kind":"pith_short_16","alias_value":"KN4PYLFXIGK57LCX","created_at":"2026-07-05T08:08:10.475884+00:00"},{"alias_kind":"pith_short_8","alias_value":"KN4PYLFX","created_at":"2026-07-05T08:08:10.475884+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.18993","citing_title":"CR-Net: Scaling Parameter-Efficient Training with Cross-Layer Low-Rank Structure","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08065","citing_title":"Deterministic Differentiable Structured Pruning for Large Language Models","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KN4PYLFXIGK57LCX22YPNBZLC6","json":"https://pith.science/pith/KN4PYLFXIGK57LCX22YPNBZLC6.json","graph_json":"https://pith.science/api/pith-number/KN4PYLFXIGK57LCX22YPNBZLC6/graph.json","events_json":"https://pith.science/api/pith-number/KN4PYLFXIGK57LCX22YPNBZLC6/events.json","paper":"https://pith.science/paper/KN4PYLFX"},"agent_actions":{"view_html":"https://pith.science/pith/KN4PYLFXIGK57LCX22YPNBZLC6","download_json":"https://pith.science/pith/KN4PYLFXIGK57LCX22YPNBZLC6.json","view_paper":"https://pith.science/paper/KN4PYLFX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.09695&json=true","fetch_graph":"https://pith.science/api/pith-number/KN4PYLFXIGK57LCX22YPNBZLC6/graph.json","fetch_events":"https://pith.science/api/pith-number/KN4PYLFXIGK57LCX22YPNBZLC6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KN4PYLFXIGK57LCX22YPNBZLC6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KN4PYLFXIGK57LCX22YPNBZLC6/action/storage_attestation","attest_author":"https://pith.science/pith/KN4PYLFXIGK57LCX22YPNBZLC6/action/author_attestation","sign_citation":"https://pith.science/pith/KN4PYLFXIGK57LCX22YPNBZLC6/action/citation_signature","submit_replication":"https://pith.science/pith/KN4PYLFXIGK57LCX22YPNBZLC6/action/replication_record"}},"created_at":"2026-07-05T08:08:10.475884+00:00","updated_at":"2026-07-05T08:08:10.475884+00:00"}