{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HWIH77GJWPGDEHE4L3UMYHV656","short_pith_number":"pith:HWIH77GJ","schema_version":"1.0","canonical_sha256":"3d907ffcc9b3cc321c9c5ee8cc1ebeef87bcec617a76deac60315fb5e9484e9c","source":{"kind":"arxiv","id":"2406.05955","version":2},"attestation_state":"computed","paper":{"title":"Turbo Sparse: Achieving LLM SOTA Performance with Minimal Activated Parameters","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Bo Wen, Haibo Chen, Haotong Xie, Li Ma, Yixin Song, Zeyu Mi, Zhengyan Zhang","submitted_at":"2024-06-10T01:21:59Z","abstract_excerpt":"Exploiting activation sparsity is a promising approach to significantly accelerating the inference process of large language models (LLMs) without compromising performance. However, activation sparsity is determined by activation functions, and commonly used ones like SwiGLU and GeGLU exhibit limited sparsity. Simply replacing these functions with ReLU fails to achieve sufficient sparsity. Moreover, inadequate training data can further increase the risk of performance degradation. To address these challenges, we propose a novel dReLU function, which is designed to improve LLM activation sparsi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05955","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-10T01:21:59Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"70176add968ceed938db7fba365cc3640a4a6db1363709cc4d67090344558c00","abstract_canon_sha256":"d55bfe05c337ea4c4a4d3f6e149f15c8dd5b9e77573ef83afc33f48397344d26"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:30:14.338805Z","signature_b64":"y6c5TLFAN2XMY6acHaP7Gmb/9/UdWN/DtdEhg8NA+mGcoDyRUiFMH8WuIDnqisdIA3QLfa1PY7Wb1qkTJhSvAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3d907ffcc9b3cc321c9c5ee8cc1ebeef87bcec617a76deac60315fb5e9484e9c","last_reissued_at":"2026-07-05T08:30:14.338354Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:30:14.338354Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Turbo Sparse: Achieving LLM SOTA Performance with Minimal Activated Parameters","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Bo Wen, Haibo Chen, Haotong Xie, Li Ma, Yixin Song, Zeyu Mi, Zhengyan Zhang","submitted_at":"2024-06-10T01:21:59Z","abstract_excerpt":"Exploiting activation sparsity is a promising approach to significantly accelerating the inference process of large language models (LLMs) without compromising performance. However, activation sparsity is determined by activation functions, and commonly used ones like SwiGLU and GeGLU exhibit limited sparsity. Simply replacing these functions with ReLU fails to achieve sufficient sparsity. Moreover, inadequate training data can further increase the risk of performance degradation. To address these challenges, we propose a novel dReLU function, which is designed to improve LLM activation sparsi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05955","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05955/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05955","created_at":"2026-07-05T08:30:14.338412+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05955v2","created_at":"2026-07-05T08:30:14.338412+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05955","created_at":"2026-07-05T08:30:14.338412+00:00"},{"alias_kind":"pith_short_12","alias_value":"HWIH77GJWPGD","created_at":"2026-07-05T08:30:14.338412+00:00"},{"alias_kind":"pith_short_16","alias_value":"HWIH77GJWPGDEHE4","created_at":"2026-07-05T08:30:14.338412+00:00"},{"alias_kind":"pith_short_8","alias_value":"HWIH77GJ","created_at":"2026-07-05T08:30:14.338412+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23191","citing_title":"Expand More, Shrink Less: Shaping Effective-Rank Dynamics for Dense Scaling in Recommendation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17659","citing_title":"Bug or Feature$^2$: Weight Drift, Activation Sparsity and Spikes","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17659","citing_title":"Bug or Feature$^2$: Weight Drift, Activation Sparsity and Spikes","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HWIH77GJWPGDEHE4L3UMYHV656","json":"https://pith.science/pith/HWIH77GJWPGDEHE4L3UMYHV656.json","graph_json":"https://pith.science/api/pith-number/HWIH77GJWPGDEHE4L3UMYHV656/graph.json","events_json":"https://pith.science/api/pith-number/HWIH77GJWPGDEHE4L3UMYHV656/events.json","paper":"https://pith.science/paper/HWIH77GJ"},"agent_actions":{"view_html":"https://pith.science/pith/HWIH77GJWPGDEHE4L3UMYHV656","download_json":"https://pith.science/pith/HWIH77GJWPGDEHE4L3UMYHV656.json","view_paper":"https://pith.science/paper/HWIH77GJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05955&json=true","fetch_graph":"https://pith.science/api/pith-number/HWIH77GJWPGDEHE4L3UMYHV656/graph.json","fetch_events":"https://pith.science/api/pith-number/HWIH77GJWPGDEHE4L3UMYHV656/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HWIH77GJWPGDEHE4L3UMYHV656/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HWIH77GJWPGDEHE4L3UMYHV656/action/storage_attestation","attest_author":"https://pith.science/pith/HWIH77GJWPGDEHE4L3UMYHV656/action/author_attestation","sign_citation":"https://pith.science/pith/HWIH77GJWPGDEHE4L3UMYHV656/action/citation_signature","submit_replication":"https://pith.science/pith/HWIH77GJWPGDEHE4L3UMYHV656/action/replication_record"}},"created_at":"2026-07-05T08:30:14.338412+00:00","updated_at":"2026-07-05T08:30:14.338412+00:00"}