{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:545EKLHDW63VPSQGBONAOB6H7Z","short_pith_number":"pith:545EKLHD","schema_version":"1.0","canonical_sha256":"ef3a452ce3b7b757ca060b9a0707c7fe628fa5f1b405d36a6bd1e2792f1a3f98","source":{"kind":"arxiv","id":"2310.04564","version":1},"attestation_state":"computed","paper":{"title":"ReLU Strikes Back: Exploiting Activation Sparsity in Large Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Carlo C Del Mundo, Golnoosh Samei, Iman Mirzadeh, Keivan Alizadeh, Mehrdad Farajtabar, Mohammad Rastegari, Oncel Tuzel, Sachin Mehta","submitted_at":"2023-10-06T20:01:33Z","abstract_excerpt":"Large Language Models (LLMs) with billions of parameters have drastically transformed AI applications. However, their demanding computation during inference has raised significant challenges for deployment on resource-constrained devices. Despite recent trends favoring alternative activation functions such as GELU or SiLU, known for increased computation, this study strongly advocates for reinstating ReLU activation in LLMs. We demonstrate that using the ReLU activation function has a negligible impact on convergence and performance while significantly reducing computation and weight transfer."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.04564","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-06T20:01:33Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ec1236753b71623ebddab14bba9dab3242399d7ed16f12e74bc801034fa41206","abstract_canon_sha256":"c7fe3ffa05d9b21382868b64cc68342cff0df6e961b0abfc4db8f6c9207cef71"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:58:20.006624Z","signature_b64":"QARGi7yY6i4tD0ZinC4MImBmu2TtcQ3Hce6AvRoyxiJFozsWqul6FbofPZlrxwOaF4AJaTmsQ3LbR3A4XnVQBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ef3a452ce3b7b757ca060b9a0707c7fe628fa5f1b405d36a6bd1e2792f1a3f98","last_reissued_at":"2026-07-05T06:58:20.006203Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:58:20.006203Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReLU Strikes Back: Exploiting Activation Sparsity in Large Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Carlo C Del Mundo, Golnoosh Samei, Iman Mirzadeh, Keivan Alizadeh, Mehrdad Farajtabar, Mohammad Rastegari, Oncel Tuzel, Sachin Mehta","submitted_at":"2023-10-06T20:01:33Z","abstract_excerpt":"Large Language Models (LLMs) with billions of parameters have drastically transformed AI applications. However, their demanding computation during inference has raised significant challenges for deployment on resource-constrained devices. Despite recent trends favoring alternative activation functions such as GELU or SiLU, known for increased computation, this study strongly advocates for reinstating ReLU activation in LLMs. We demonstrate that using the ReLU activation function has a negligible impact on convergence and performance while significantly reducing computation and weight transfer."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.04564","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.04564/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.04564","created_at":"2026-07-05T06:58:20.006257+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.04564v1","created_at":"2026-07-05T06:58:20.006257+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.04564","created_at":"2026-07-05T06:58:20.006257+00:00"},{"alias_kind":"pith_short_12","alias_value":"545EKLHDW63V","created_at":"2026-07-05T06:58:20.006257+00:00"},{"alias_kind":"pith_short_16","alias_value":"545EKLHDW63VPSQG","created_at":"2026-07-05T06:58:20.006257+00:00"},{"alias_kind":"pith_short_8","alias_value":"545EKLHD","created_at":"2026-07-05T06:58:20.006257+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23419","citing_title":"GRINQH: Graded Input-based Quantization Hierarchy for Efficient LLM Generation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07495","citing_title":"Second-Order Path Kernel Interpolation Formulas in Machine Learning","ref_index":162,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25704","citing_title":"PowLU: An Activation Function for Stable Pre-Training of LLMs","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26632","citing_title":"RT-Lynx: Putting GEMM Sparsity in the Right Place for Diffusion Models","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12744","citing_title":"Resting Neurons, Active Insights: Robustifying Activation Sparsity in LLMs via Spontaneity","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2506.06374","citing_title":"SiLIF: Structured State Space Model Dynamics and Parametrization for Spiking Neural Networks","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22166","citing_title":"Motivating Next-Gen Accelerators with Flexible (N:M) Activation Sparsity via Benchmarking Lightweight Post-Training Sparsification Approaches","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12744","citing_title":"Resting Neurons, Active Insights: Robustifying Activation Sparsity in LLMs via Spontaneity","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/545EKLHDW63VPSQGBONAOB6H7Z","json":"https://pith.science/pith/545EKLHDW63VPSQGBONAOB6H7Z.json","graph_json":"https://pith.science/api/pith-number/545EKLHDW63VPSQGBONAOB6H7Z/graph.json","events_json":"https://pith.science/api/pith-number/545EKLHDW63VPSQGBONAOB6H7Z/events.json","paper":"https://pith.science/paper/545EKLHD"},"agent_actions":{"view_html":"https://pith.science/pith/545EKLHDW63VPSQGBONAOB6H7Z","download_json":"https://pith.science/pith/545EKLHDW63VPSQGBONAOB6H7Z.json","view_paper":"https://pith.science/paper/545EKLHD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.04564&json=true","fetch_graph":"https://pith.science/api/pith-number/545EKLHDW63VPSQGBONAOB6H7Z/graph.json","fetch_events":"https://pith.science/api/pith-number/545EKLHDW63VPSQGBONAOB6H7Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/545EKLHDW63VPSQGBONAOB6H7Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/545EKLHDW63VPSQGBONAOB6H7Z/action/storage_attestation","attest_author":"https://pith.science/pith/545EKLHDW63VPSQGBONAOB6H7Z/action/author_attestation","sign_citation":"https://pith.science/pith/545EKLHDW63VPSQGBONAOB6H7Z/action/citation_signature","submit_replication":"https://pith.science/pith/545EKLHDW63VPSQGBONAOB6H7Z/action/replication_record"}},"created_at":"2026-07-05T06:58:20.006257+00:00","updated_at":"2026-07-05T06:58:20.006257+00:00"}