{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JAO7KWUSRS7JEDUDKL5TXDGQ44","short_pith_number":"pith:JAO7KWUS","schema_version":"1.0","canonical_sha256":"481df55a928cbe920e8352fb3b8cd0e71cb4db0140c204509f4a9a91e1148fba","source":{"kind":"arxiv","id":"2505.17974","version":1},"attestation_state":"computed","paper":{"title":"Generalized Fisher-Weighted SVD: Scalable Kronecker-Factored Fisher Approximation for Compressing Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Andrey Kuznetsov, Daniil Moskovskiy, Daria Cherniuk, Evgeny Frolov, Maxim Kurkin, Viktoriia Chekalina","submitted_at":"2025-05-23T14:41:52Z","abstract_excerpt":"The Fisher information is a fundamental concept for characterizing the sensitivity of parameters in neural networks. However, leveraging the full observed Fisher information is too expensive for large models, so most methods rely on simple diagonal approximations. While efficient, this approach ignores parameter correlations, often resulting in reduced performance on downstream tasks. In this work, we mitigate these limitations and propose Generalized Fisher-Weighted SVD (GFWSVD), a post-training LLM compression technique that accounts for both diagonal and off-diagonal elements of the Fisher "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.17974","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-23T14:41:52Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"09d8c82032ef33c53eb9d804b962248076b26be7b69e9ada9925fa5c77c62c73","abstract_canon_sha256":"678a4d8513e049d8f7218be1ba048be5e4580af1accac9930ce27805b9f33217"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:34.559572Z","signature_b64":"jd6T84eQRM4qkd0yhG4khf/FkXfsFjDstIzjues6rh4NiTaDxCLtoBIUipEWOvqyQnIwgC7GbYymA4ZCLad9Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"481df55a928cbe920e8352fb3b8cd0e71cb4db0140c204509f4a9a91e1148fba","last_reissued_at":"2026-07-05T11:08:34.559033Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:34.559033Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generalized Fisher-Weighted SVD: Scalable Kronecker-Factored Fisher Approximation for Compressing Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Andrey Kuznetsov, Daniil Moskovskiy, Daria Cherniuk, Evgeny Frolov, Maxim Kurkin, Viktoriia Chekalina","submitted_at":"2025-05-23T14:41:52Z","abstract_excerpt":"The Fisher information is a fundamental concept for characterizing the sensitivity of parameters in neural networks. However, leveraging the full observed Fisher information is too expensive for large models, so most methods rely on simple diagonal approximations. While efficient, this approach ignores parameter correlations, often resulting in reduced performance on downstream tasks. In this work, we mitigate these limitations and propose Generalized Fisher-Weighted SVD (GFWSVD), a post-training LLM compression technique that accounts for both diagonal and off-diagonal elements of the Fisher "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.17974","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.17974/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.17974","created_at":"2026-07-05T11:08:34.559095+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.17974v1","created_at":"2026-07-05T11:08:34.559095+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.17974","created_at":"2026-07-05T11:08:34.559095+00:00"},{"alias_kind":"pith_short_12","alias_value":"JAO7KWUSRS7J","created_at":"2026-07-05T11:08:34.559095+00:00"},{"alias_kind":"pith_short_16","alias_value":"JAO7KWUSRS7JEDUD","created_at":"2026-07-05T11:08:34.559095+00:00"},{"alias_kind":"pith_short_8","alias_value":"JAO7KWUS","created_at":"2026-07-05T11:08:34.559095+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":3,"sample":[{"citing_arxiv_id":"2606.19993","citing_title":"Activation- and Influence-Aware Ranks (AIR): Function-Preserving SVD Compression for LLMs","ref_index":171,"is_internal_anchor":true},{"citing_arxiv_id":"2605.15626","citing_title":"IO-SVD: Input-Output Whitened SVD for Adaptive-Rank LLM Compression","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2605.08314","citing_title":"FlashSVD v1.5: Making Low-Rank Transformers Inference Actually Fast","ref_index":16,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JAO7KWUSRS7JEDUDKL5TXDGQ44","json":"https://pith.science/pith/JAO7KWUSRS7JEDUDKL5TXDGQ44.json","graph_json":"https://pith.science/api/pith-number/JAO7KWUSRS7JEDUDKL5TXDGQ44/graph.json","events_json":"https://pith.science/api/pith-number/JAO7KWUSRS7JEDUDKL5TXDGQ44/events.json","paper":"https://pith.science/paper/JAO7KWUS"},"agent_actions":{"view_html":"https://pith.science/pith/JAO7KWUSRS7JEDUDKL5TXDGQ44","download_json":"https://pith.science/pith/JAO7KWUSRS7JEDUDKL5TXDGQ44.json","view_paper":"https://pith.science/paper/JAO7KWUS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.17974&json=true","fetch_graph":"https://pith.science/api/pith-number/JAO7KWUSRS7JEDUDKL5TXDGQ44/graph.json","fetch_events":"https://pith.science/api/pith-number/JAO7KWUSRS7JEDUDKL5TXDGQ44/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JAO7KWUSRS7JEDUDKL5TXDGQ44/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JAO7KWUSRS7JEDUDKL5TXDGQ44/action/storage_attestation","attest_author":"https://pith.science/pith/JAO7KWUSRS7JEDUDKL5TXDGQ44/action/author_attestation","sign_citation":"https://pith.science/pith/JAO7KWUSRS7JEDUDKL5TXDGQ44/action/citation_signature","submit_replication":"https://pith.science/pith/JAO7KWUSRS7JEDUDKL5TXDGQ44/action/replication_record"}},"created_at":"2026-07-05T11:08:34.559095+00:00","updated_at":"2026-07-05T11:08:34.559095+00:00"}