{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:J3AKP7JC727OPLP32FPZ7IHY2Y","short_pith_number":"pith:J3AKP7JC","schema_version":"1.0","canonical_sha256":"4ec0a7fd22febee7adfbd15f9fa0f8d60c123aea446eb472b384dbf6649c45ab","source":{"kind":"arxiv","id":"2406.06562","version":1},"attestation_state":"computed","paper":{"title":"Achieving Sparse Activation in Small Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Boyuan Yang, Jifeng Song, Kai Huang, Wei Gao, Xiangyu Yin","submitted_at":"2024-06-03T03:21:49Z","abstract_excerpt":"Sparse activation, which selectively activates only an input-dependent set of neurons in inference, is a useful technique to reduce the computing cost of Large Language Models (LLMs) without retraining or adaptation efforts. However, whether it can be applied to the recently emerging Small Language Models (SLMs) remains questionable, because SLMs are generally less over-parameterized than LLMs. In this paper, we aim to achieve sparse activation in SLMs. We first show that the existing sparse activation schemes in LLMs that build on neurons' output magnitudes cannot be applied to SLMs, and acti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.06562","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-03T03:21:49Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e9e83c9659f55d97f4d967af6bf645c304173a6c1d31905ad375786a40d0f508","abstract_canon_sha256":"c3221c5aab270e0e5504ff088d7a67c520163e07ae4ffcecbe057808dcc9c33a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:29:47.455474Z","signature_b64":"JF/lGAeyL0MpIPg8Fvc20gXFe26ySoTzjsPlrN368bhc1hCoPFXCfl3OdcH6Ecwoeqeo9zKk1AJY89yfCiBJBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4ec0a7fd22febee7adfbd15f9fa0f8d60c123aea446eb472b384dbf6649c45ab","last_reissued_at":"2026-07-05T08:29:47.455116Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:29:47.455116Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Achieving Sparse Activation in Small Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Boyuan Yang, Jifeng Song, Kai Huang, Wei Gao, Xiangyu Yin","submitted_at":"2024-06-03T03:21:49Z","abstract_excerpt":"Sparse activation, which selectively activates only an input-dependent set of neurons in inference, is a useful technique to reduce the computing cost of Large Language Models (LLMs) without retraining or adaptation efforts. However, whether it can be applied to the recently emerging Small Language Models (SLMs) remains questionable, because SLMs are generally less over-parameterized than LLMs. In this paper, we aim to achieve sparse activation in SLMs. We first show that the existing sparse activation schemes in LLMs that build on neurons' output magnitudes cannot be applied to SLMs, and acti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.06562","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.06562/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.06562","created_at":"2026-07-05T08:29:47.455171+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.06562v1","created_at":"2026-07-05T08:29:47.455171+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.06562","created_at":"2026-07-05T08:29:47.455171+00:00"},{"alias_kind":"pith_short_12","alias_value":"J3AKP7JC727O","created_at":"2026-07-05T08:29:47.455171+00:00"},{"alias_kind":"pith_short_16","alias_value":"J3AKP7JC727OPLP3","created_at":"2026-07-05T08:29:47.455171+00:00"},{"alias_kind":"pith_short_8","alias_value":"J3AKP7JC","created_at":"2026-07-05T08:29:47.455171+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J3AKP7JC727OPLP32FPZ7IHY2Y","json":"https://pith.science/pith/J3AKP7JC727OPLP32FPZ7IHY2Y.json","graph_json":"https://pith.science/api/pith-number/J3AKP7JC727OPLP32FPZ7IHY2Y/graph.json","events_json":"https://pith.science/api/pith-number/J3AKP7JC727OPLP32FPZ7IHY2Y/events.json","paper":"https://pith.science/paper/J3AKP7JC"},"agent_actions":{"view_html":"https://pith.science/pith/J3AKP7JC727OPLP32FPZ7IHY2Y","download_json":"https://pith.science/pith/J3AKP7JC727OPLP32FPZ7IHY2Y.json","view_paper":"https://pith.science/paper/J3AKP7JC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.06562&json=true","fetch_graph":"https://pith.science/api/pith-number/J3AKP7JC727OPLP32FPZ7IHY2Y/graph.json","fetch_events":"https://pith.science/api/pith-number/J3AKP7JC727OPLP32FPZ7IHY2Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J3AKP7JC727OPLP32FPZ7IHY2Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J3AKP7JC727OPLP32FPZ7IHY2Y/action/storage_attestation","attest_author":"https://pith.science/pith/J3AKP7JC727OPLP32FPZ7IHY2Y/action/author_attestation","sign_citation":"https://pith.science/pith/J3AKP7JC727OPLP32FPZ7IHY2Y/action/citation_signature","submit_replication":"https://pith.science/pith/J3AKP7JC727OPLP32FPZ7IHY2Y/action/replication_record"}},"created_at":"2026-07-05T08:29:47.455171+00:00","updated_at":"2026-07-05T08:29:47.455171+00:00"}