{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:7XGW6WXWN4KEI6HI6NBVVWDTGG","short_pith_number":"pith:7XGW6WXW","schema_version":"1.0","canonical_sha256":"fdcd6f5af66f144478e8f3435ad8733195508099fd5a7cc46f0b0176e36593d7","source":{"kind":"arxiv","id":"2106.04426","version":3},"attestation_state":"computed","paper":{"title":"Hash Layers For Large Sparse Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Arthur Szlam, Jason Weston, Sainbayar Sukhbaatar, Stephen Roller","submitted_at":"2021-06-08T14:54:24Z","abstract_excerpt":"We investigate the training of sparse layers that use different parameters for different inputs based on hashing in large Transformer models. Specifically, we modify the feedforward layer to hash to different sets of weights depending on the current token, over all tokens in the sequence. We show that this procedure either outperforms or is competitive with learning-to-route mixture-of-expert methods such as Switch Transformers and BASE Layers, while requiring no routing parameters or extra terms in the objective function such as a load balancing loss, and no sophisticated assignment algorithm"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.04426","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-08T14:54:24Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"d101f2a10c16896dade654f80cf7b57ae78c3c9c7584ded00cec664dd13f1664","abstract_canon_sha256":"f5ed9ee3a1979dd3bf7b019fb3a5dd716ee5df4b29d223e0b5d8353e26337c05"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:59:22.787147Z","signature_b64":"/d2vftAtSYbj1xfpBvZW1Ox5x61tAmNQ3lJydqbRWLFHMh+i71YPTzxhsUwcbmF5jQWnwUjuBT5tJqnM+YkQBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fdcd6f5af66f144478e8f3435ad8733195508099fd5a7cc46f0b0176e36593d7","last_reissued_at":"2026-07-05T02:59:22.786716Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:59:22.786716Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hash Layers For Large Sparse Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Arthur Szlam, Jason Weston, Sainbayar Sukhbaatar, Stephen Roller","submitted_at":"2021-06-08T14:54:24Z","abstract_excerpt":"We investigate the training of sparse layers that use different parameters for different inputs based on hashing in large Transformer models. Specifically, we modify the feedforward layer to hash to different sets of weights depending on the current token, over all tokens in the sequence. We show that this procedure either outperforms or is competitive with learning-to-route mixture-of-expert methods such as Switch Transformers and BASE Layers, while requiring no routing parameters or extra terms in the objective function such as a load balancing loss, and no sophisticated assignment algorithm"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.04426","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.04426/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.04426","created_at":"2026-07-05T02:59:22.786776+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.04426v3","created_at":"2026-07-05T02:59:22.786776+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.04426","created_at":"2026-07-05T02:59:22.786776+00:00"},{"alias_kind":"pith_short_12","alias_value":"7XGW6WXWN4KE","created_at":"2026-07-05T02:59:22.786776+00:00"},{"alias_kind":"pith_short_16","alias_value":"7XGW6WXWN4KEI6HI","created_at":"2026-07-05T02:59:22.786776+00:00"},{"alias_kind":"pith_short_8","alias_value":"7XGW6WXW","created_at":"2026-07-05T02:59:22.786776+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17743","citing_title":"MoASE++: Mixture of Activation Sparsity Experts with Domain-Adaptive On-policy Distillation for Continual Test Time Adaptation","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2202.08906","citing_title":"ST-MoE: Designing Stable and Transferable Sparse Expert Models","ref_index":190,"is_internal_anchor":false},{"citing_arxiv_id":"2401.06066","citing_title":"DeepSeekMoE: Towards Ultimate Expert Specialization in Mixture-of-Experts Language Models","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7XGW6WXWN4KEI6HI6NBVVWDTGG","json":"https://pith.science/pith/7XGW6WXWN4KEI6HI6NBVVWDTGG.json","graph_json":"https://pith.science/api/pith-number/7XGW6WXWN4KEI6HI6NBVVWDTGG/graph.json","events_json":"https://pith.science/api/pith-number/7XGW6WXWN4KEI6HI6NBVVWDTGG/events.json","paper":"https://pith.science/paper/7XGW6WXW"},"agent_actions":{"view_html":"https://pith.science/pith/7XGW6WXWN4KEI6HI6NBVVWDTGG","download_json":"https://pith.science/pith/7XGW6WXWN4KEI6HI6NBVVWDTGG.json","view_paper":"https://pith.science/paper/7XGW6WXW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.04426&json=true","fetch_graph":"https://pith.science/api/pith-number/7XGW6WXWN4KEI6HI6NBVVWDTGG/graph.json","fetch_events":"https://pith.science/api/pith-number/7XGW6WXWN4KEI6HI6NBVVWDTGG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7XGW6WXWN4KEI6HI6NBVVWDTGG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7XGW6WXWN4KEI6HI6NBVVWDTGG/action/storage_attestation","attest_author":"https://pith.science/pith/7XGW6WXWN4KEI6HI6NBVVWDTGG/action/author_attestation","sign_citation":"https://pith.science/pith/7XGW6WXWN4KEI6HI6NBVVWDTGG/action/citation_signature","submit_replication":"https://pith.science/pith/7XGW6WXWN4KEI6HI6NBVVWDTGG/action/replication_record"}},"created_at":"2026-07-05T02:59:22.786776+00:00","updated_at":"2026-07-05T02:59:22.786776+00:00"}