{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:X3GJ3PWSFOSNELLXBR2QID734U","short_pith_number":"pith:X3GJ3PWS","schema_version":"1.0","canonical_sha256":"becc9dbed22ba4d22d770c75040ffbe5031916e93c490d80125089bbb679599e","source":{"kind":"arxiv","id":"2501.12370","version":3},"attestation_state":"computed","paper":{"title":"Parameters vs FLOPs: Scaling Laws for Optimal Sparsity for Mixture-of-Experts Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alaaeldin Mohamed Elnouby Ali, Dan Busbridge, Harshay Shah, Josh Susskind, Samira Abnar, Vimal Thilak","submitted_at":"2025-01-21T18:51:15Z","abstract_excerpt":"Scaling the capacity of language models has consistently proven to be a reliable approach for improving performance and unlocking new capabilities. Capacity can be primarily defined by two dimensions: the number of model parameters and the compute per example. While scaling typically involves increasing both, the precise interplay between these factors and their combined contribution to overall capacity remains not fully understood. We explore this relationship in the context of sparse Mixture-of-Experts (MoEs), which allow scaling the number of parameters without proportionally increasing the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.12370","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-01-21T18:51:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fafd60d2820e6e90d3807fbde95e8d831aee4bf3a28771a1450a2315e8b5633e","abstract_canon_sha256":"571d784b0cc8d787d5e6bf3daa601f97a91d55fc96ef7e12dd9679e4dc38bc37"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:31:04.462743Z","signature_b64":"Rysr/LBneXgbcmBI53VPebfoYkvva5Q8jKfhhLRaJPh1XGMgKu/Ck3SyrMY21VqI3KOp+9MVzDutrvE10C/kCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"becc9dbed22ba4d22d770c75040ffbe5031916e93c490d80125089bbb679599e","last_reissued_at":"2026-07-05T11:31:04.462241Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:31:04.462241Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Parameters vs FLOPs: Scaling Laws for Optimal Sparsity for Mixture-of-Experts Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alaaeldin Mohamed Elnouby Ali, Dan Busbridge, Harshay Shah, Josh Susskind, Samira Abnar, Vimal Thilak","submitted_at":"2025-01-21T18:51:15Z","abstract_excerpt":"Scaling the capacity of language models has consistently proven to be a reliable approach for improving performance and unlocking new capabilities. Capacity can be primarily defined by two dimensions: the number of model parameters and the compute per example. While scaling typically involves increasing both, the precise interplay between these factors and their combined contribution to overall capacity remains not fully understood. We explore this relationship in the context of sparse Mixture-of-Experts (MoEs), which allow scaling the number of parameters without proportionally increasing the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.12370","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.12370/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.12370","created_at":"2026-07-05T11:31:04.462299+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.12370v3","created_at":"2026-07-05T11:31:04.462299+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.12370","created_at":"2026-07-05T11:31:04.462299+00:00"},{"alias_kind":"pith_short_12","alias_value":"X3GJ3PWSFOSN","created_at":"2026-07-05T11:31:04.462299+00:00"},{"alias_kind":"pith_short_16","alias_value":"X3GJ3PWSFOSNELLX","created_at":"2026-07-05T11:31:04.462299+00:00"},{"alias_kind":"pith_short_8","alias_value":"X3GJ3PWS","created_at":"2026-07-05T11:31:04.462299+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.15079","citing_title":"Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01062","citing_title":"DAG-MoE: From Simple Mixture to Structural Aggregation in Mixture-of-Experts","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2506.12119","citing_title":"Mixture-of-Experts Can Surpass Dense LLMs Under Strictly Equal Resource","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18245","citing_title":"Scaling Laws Meet Model Architecture: Toward Inference-Efficient LLMs","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19835","citing_title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19835","citing_title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X3GJ3PWSFOSNELLXBR2QID734U","json":"https://pith.science/pith/X3GJ3PWSFOSNELLXBR2QID734U.json","graph_json":"https://pith.science/api/pith-number/X3GJ3PWSFOSNELLXBR2QID734U/graph.json","events_json":"https://pith.science/api/pith-number/X3GJ3PWSFOSNELLXBR2QID734U/events.json","paper":"https://pith.science/paper/X3GJ3PWS"},"agent_actions":{"view_html":"https://pith.science/pith/X3GJ3PWSFOSNELLXBR2QID734U","download_json":"https://pith.science/pith/X3GJ3PWSFOSNELLXBR2QID734U.json","view_paper":"https://pith.science/paper/X3GJ3PWS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.12370&json=true","fetch_graph":"https://pith.science/api/pith-number/X3GJ3PWSFOSNELLXBR2QID734U/graph.json","fetch_events":"https://pith.science/api/pith-number/X3GJ3PWSFOSNELLXBR2QID734U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X3GJ3PWSFOSNELLXBR2QID734U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X3GJ3PWSFOSNELLXBR2QID734U/action/storage_attestation","attest_author":"https://pith.science/pith/X3GJ3PWSFOSNELLXBR2QID734U/action/author_attestation","sign_citation":"https://pith.science/pith/X3GJ3PWSFOSNELLXBR2QID734U/action/citation_signature","submit_replication":"https://pith.science/pith/X3GJ3PWSFOSNELLXBR2QID734U/action/replication_record"}},"created_at":"2026-07-05T11:31:04.462299+00:00","updated_at":"2026-07-05T11:31:04.462299+00:00"}