{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:SMGHUSWEQO74QFINNGZDIT2YNS","short_pith_number":"pith:SMGHUSWE","schema_version":"1.0","canonical_sha256":"930c7a4ac483bfc8150d69b2344f586c86ab89459393228bd651a68fd6e3c0e9","source":{"kind":"arxiv","id":"2104.06022","version":4},"attestation_state":"computed","paper":{"title":"Lessons on Parameter Sharing across Layers in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Sho Takase, Shun Kiyono","submitted_at":"2021-04-13T08:41:07Z","abstract_excerpt":"We propose a parameter sharing method for Transformers (Vaswani et al., 2017). The proposed approach relaxes a widely used technique, which shares parameters for one layer with all layers such as Universal Transformers (Dehghani et al., 2019), to increase the efficiency in the computational time. We propose three strategies: Sequence, Cycle, and Cycle (rev) to assign parameters to each layer. Experimental results show that the proposed strategies are efficient in the parameter size and computational time. Moreover, we indicate that the proposed strategies are also effective in the configuratio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.06022","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-04-13T08:41:07Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"98786192191147e385beb341669018284b7572dcdf4ef7c5160e13f6843456e9","abstract_canon_sha256":"4b34d1946866202fd774d48d9f376e7b126ccbae11004f229863cc35054524ae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:16:44.590859Z","signature_b64":"Z5LkMhUqqhBfqhNCnb/KMlMRP7Ehbbuh9xfBYMdkM0R/Im790Bmrp4To/BAwBd5QjpiePwUV73sthtzGCBVlCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"930c7a4ac483bfc8150d69b2344f586c86ab89459393228bd651a68fd6e3c0e9","last_reissued_at":"2026-07-05T06:16:44.590339Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:16:44.590339Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Lessons on Parameter Sharing across Layers in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Sho Takase, Shun Kiyono","submitted_at":"2021-04-13T08:41:07Z","abstract_excerpt":"We propose a parameter sharing method for Transformers (Vaswani et al., 2017). The proposed approach relaxes a widely used technique, which shares parameters for one layer with all layers such as Universal Transformers (Dehghani et al., 2019), to increase the efficiency in the computational time. We propose three strategies: Sequence, Cycle, and Cycle (rev) to assign parameters to each layer. Experimental results show that the proposed strategies are efficient in the parameter size and computational time. Moreover, we indicate that the proposed strategies are also effective in the configuratio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.06022","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.06022/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.06022","created_at":"2026-07-05T06:16:44.590402+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.06022v4","created_at":"2026-07-05T06:16:44.590402+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.06022","created_at":"2026-07-05T06:16:44.590402+00:00"},{"alias_kind":"pith_short_12","alias_value":"SMGHUSWEQO74","created_at":"2026-07-05T06:16:44.590402+00:00"},{"alias_kind":"pith_short_16","alias_value":"SMGHUSWEQO74QFIN","created_at":"2026-07-05T06:16:44.590402+00:00"},{"alias_kind":"pith_short_8","alias_value":"SMGHUSWE","created_at":"2026-07-05T06:16:44.590402+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20737","citing_title":"Repeated Shared Access Enables Grokking, but Edit Propagation Depends on an Addressable Memory","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11860","citing_title":"RePAIR: Predictive Self-Supervised Representation Learning in Chess","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2510.25741","citing_title":"Scaling Latent Reasoning via Looped Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05171","citing_title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","ref_index":153,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SMGHUSWEQO74QFINNGZDIT2YNS","json":"https://pith.science/pith/SMGHUSWEQO74QFINNGZDIT2YNS.json","graph_json":"https://pith.science/api/pith-number/SMGHUSWEQO74QFINNGZDIT2YNS/graph.json","events_json":"https://pith.science/api/pith-number/SMGHUSWEQO74QFINNGZDIT2YNS/events.json","paper":"https://pith.science/paper/SMGHUSWE"},"agent_actions":{"view_html":"https://pith.science/pith/SMGHUSWEQO74QFINNGZDIT2YNS","download_json":"https://pith.science/pith/SMGHUSWEQO74QFINNGZDIT2YNS.json","view_paper":"https://pith.science/paper/SMGHUSWE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.06022&json=true","fetch_graph":"https://pith.science/api/pith-number/SMGHUSWEQO74QFINNGZDIT2YNS/graph.json","fetch_events":"https://pith.science/api/pith-number/SMGHUSWEQO74QFINNGZDIT2YNS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SMGHUSWEQO74QFINNGZDIT2YNS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SMGHUSWEQO74QFINNGZDIT2YNS/action/storage_attestation","attest_author":"https://pith.science/pith/SMGHUSWEQO74QFINNGZDIT2YNS/action/author_attestation","sign_citation":"https://pith.science/pith/SMGHUSWEQO74QFINNGZDIT2YNS/action/citation_signature","submit_replication":"https://pith.science/pith/SMGHUSWEQO74QFINNGZDIT2YNS/action/replication_record"}},"created_at":"2026-07-05T06:16:44.590402+00:00","updated_at":"2026-07-05T06:16:44.590402+00:00"}