{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RCOHOGZFTSHK5RV3HQYCJKWHWN","short_pith_number":"pith:RCOHOGZF","schema_version":"1.0","canonical_sha256":"889c771b259c8eaec6bb3c3024aac7b36608937bfb918ee8be0eeed85170983c","source":{"kind":"arxiv","id":"2404.08763","version":4},"attestation_state":"computed","paper":{"title":"CATS: Contextually-Aware Thresholding for Sparsity in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Azalia Mirhoseini, Donghyun Lee, Genghan Zhang, Je-Yong Lee, Mo Tiwari","submitted_at":"2024-04-12T18:42:18Z","abstract_excerpt":"Large Language Models (LLMs) have dramatically advanced AI applications, yet their deployment remains challenging due to their immense inference costs. Recent studies ameliorate the computational costs of LLMs by increasing their activation sparsity but suffer from significant performance degradation on downstream tasks. In this work, we introduce a new framework for sparsifying the activations of base LLMs and reducing inference costs, dubbed Contextually Aware Thresholding for Sparsity (CATS). CATS is relatively simple, easy to implement, and highly effective. At the heart of our framework i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.08763","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-12T18:42:18Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"6150d44ba710b6f5665b33879f926de67828ce4a84625b6dd186a256937ab6a2","abstract_canon_sha256":"9035781d739aff58acffbca363960a746cafba0e9997e2525ec2a1f44c21841b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:30:12.423228Z","signature_b64":"Fcjq3RWwVOnQXk6Shx0UL4uglyG/Y8HuphbEilqKxhGnzViIVWLj7tMoFOefPeVJf7qL6phXHUt2xE/GvmYoAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"889c771b259c8eaec6bb3c3024aac7b36608937bfb918ee8be0eeed85170983c","last_reissued_at":"2026-07-05T09:30:12.422808Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:30:12.422808Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CATS: Contextually-Aware Thresholding for Sparsity in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Azalia Mirhoseini, Donghyun Lee, Genghan Zhang, Je-Yong Lee, Mo Tiwari","submitted_at":"2024-04-12T18:42:18Z","abstract_excerpt":"Large Language Models (LLMs) have dramatically advanced AI applications, yet their deployment remains challenging due to their immense inference costs. Recent studies ameliorate the computational costs of LLMs by increasing their activation sparsity but suffer from significant performance degradation on downstream tasks. In this work, we introduce a new framework for sparsifying the activations of base LLMs and reducing inference costs, dubbed Contextually Aware Thresholding for Sparsity (CATS). CATS is relatively simple, easy to implement, and highly effective. At the heart of our framework i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.08763","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.08763/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.08763","created_at":"2026-07-05T09:30:12.422864+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.08763v4","created_at":"2026-07-05T09:30:12.422864+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.08763","created_at":"2026-07-05T09:30:12.422864+00:00"},{"alias_kind":"pith_short_12","alias_value":"RCOHOGZFTSHK","created_at":"2026-07-05T09:30:12.422864+00:00"},{"alias_kind":"pith_short_16","alias_value":"RCOHOGZFTSHK5RV3","created_at":"2026-07-05T09:30:12.422864+00:00"},{"alias_kind":"pith_short_8","alias_value":"RCOHOGZF","created_at":"2026-07-05T09:30:12.422864+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26632","citing_title":"RT-Lynx: Putting the GEMM Sparsity In a Right Way for Diffusion Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12744","citing_title":"Resting Neurons, Active Insights: Robustifying Activation Sparsity in LLMs via Spontaneity","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17659","citing_title":"Bug or Feature$^2$: Weight Drift, Activation Sparsity and Spikes","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17659","citing_title":"Bug or Feature$^2$: Weight Drift, Activation Sparsity and Spikes","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22166","citing_title":"Motivating Next-Gen Accelerators with Flexible (N:M) Activation Sparsity via Benchmarking Lightweight Post-Training Sparsification Approaches","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12744","citing_title":"Resting Neurons, Active Insights: Robustifying Activation Sparsity in LLMs via Spontaneity","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05899","citing_title":"VisMMOE: Exploiting Visual-Expert Affinity for Efficient Visual-Language MoE Offloading","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RCOHOGZFTSHK5RV3HQYCJKWHWN","json":"https://pith.science/pith/RCOHOGZFTSHK5RV3HQYCJKWHWN.json","graph_json":"https://pith.science/api/pith-number/RCOHOGZFTSHK5RV3HQYCJKWHWN/graph.json","events_json":"https://pith.science/api/pith-number/RCOHOGZFTSHK5RV3HQYCJKWHWN/events.json","paper":"https://pith.science/paper/RCOHOGZF"},"agent_actions":{"view_html":"https://pith.science/pith/RCOHOGZFTSHK5RV3HQYCJKWHWN","download_json":"https://pith.science/pith/RCOHOGZFTSHK5RV3HQYCJKWHWN.json","view_paper":"https://pith.science/paper/RCOHOGZF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.08763&json=true","fetch_graph":"https://pith.science/api/pith-number/RCOHOGZFTSHK5RV3HQYCJKWHWN/graph.json","fetch_events":"https://pith.science/api/pith-number/RCOHOGZFTSHK5RV3HQYCJKWHWN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RCOHOGZFTSHK5RV3HQYCJKWHWN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RCOHOGZFTSHK5RV3HQYCJKWHWN/action/storage_attestation","attest_author":"https://pith.science/pith/RCOHOGZFTSHK5RV3HQYCJKWHWN/action/author_attestation","sign_citation":"https://pith.science/pith/RCOHOGZFTSHK5RV3HQYCJKWHWN/action/citation_signature","submit_replication":"https://pith.science/pith/RCOHOGZFTSHK5RV3HQYCJKWHWN/action/replication_record"}},"created_at":"2026-07-05T09:30:12.422864+00:00","updated_at":"2026-07-05T09:30:12.422864+00:00"}