{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QP6QWOJC7JJCDMVJZPVS7CJ55L","short_pith_number":"pith:QP6QWOJC","schema_version":"1.0","canonical_sha256":"83fd0b3922fa5221b2a9cbeb2f893deaebd285e3db8b0d939bd26b9723d77704","source":{"kind":"arxiv","id":"2412.05644","version":3},"attestation_state":"computed","paper":{"title":"Mixture of Hidden-Dimensions Transformer","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"HaiFeng Wang, Hua Wu, Jiawei Sheng, Junyuan Shang, Shuohuan Wang, Tingwen Liu, Yilong Chen, Yu Sun, Zhengyu Zhang","submitted_at":"2024-12-07T13:15:22Z","abstract_excerpt":"Transformer models encounter challenges in scaling hidden dimensions efficiently, as uniformly increasing them inflates computational and memory costs while failing to emphasize the most relevant features for each token. For further understanding, we study hidden dimension sparsity and observe that trained Transformers utilize only a small fraction of token dimensions, revealing an \"activation flow\" pattern. Notably, there are shared sub-dimensions with sustained activation across multiple consecutive tokens and specialized sub-dimensions uniquely activated for each token. To better model toke"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.05644","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-07T13:15:22Z","cross_cats_sorted":[],"title_canon_sha256":"ce364de6537412fd1d3a22bf9df15197dcd57379402fea2cfed78ff1617bbe92","abstract_canon_sha256":"29728c518502cb2ad5fae047c293f959696c8d4153a48f2c405c880ff4492331"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:49:47.206691Z","signature_b64":"3P+w/azSdQL900xJDz48sDZ9+tGf0gB+Xi6lcbWq7VgjKrOY9EHxj+OmQ9eDkqVjw7uYel9d/SXoY5vLX7r2AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"83fd0b3922fa5221b2a9cbeb2f893deaebd285e3db8b0d939bd26b9723d77704","last_reissued_at":"2026-07-05T09:49:47.206200Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:49:47.206200Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mixture of Hidden-Dimensions Transformer","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"HaiFeng Wang, Hua Wu, Jiawei Sheng, Junyuan Shang, Shuohuan Wang, Tingwen Liu, Yilong Chen, Yu Sun, Zhengyu Zhang","submitted_at":"2024-12-07T13:15:22Z","abstract_excerpt":"Transformer models encounter challenges in scaling hidden dimensions efficiently, as uniformly increasing them inflates computational and memory costs while failing to emphasize the most relevant features for each token. For further understanding, we study hidden dimension sparsity and observe that trained Transformers utilize only a small fraction of token dimensions, revealing an \"activation flow\" pattern. Notably, there are shared sub-dimensions with sustained activation across multiple consecutive tokens and specialized sub-dimensions uniquely activated for each token. To better model toke"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.05644","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.05644/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.05644","created_at":"2026-07-05T09:49:47.206262+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.05644v3","created_at":"2026-07-05T09:49:47.206262+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.05644","created_at":"2026-07-05T09:49:47.206262+00:00"},{"alias_kind":"pith_short_12","alias_value":"QP6QWOJC7JJC","created_at":"2026-07-05T09:49:47.206262+00:00"},{"alias_kind":"pith_short_16","alias_value":"QP6QWOJC7JJCDMVJ","created_at":"2026-07-05T09:49:47.206262+00:00"},{"alias_kind":"pith_short_8","alias_value":"QP6QWOJC","created_at":"2026-07-05T09:49:47.206262+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.06844","citing_title":"Adapt Once, Thrive with Updates: Transferable Parameter-Efficient Fine-Tuning on Evolving Base Models","ref_index":7,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QP6QWOJC7JJCDMVJZPVS7CJ55L","json":"https://pith.science/pith/QP6QWOJC7JJCDMVJZPVS7CJ55L.json","graph_json":"https://pith.science/api/pith-number/QP6QWOJC7JJCDMVJZPVS7CJ55L/graph.json","events_json":"https://pith.science/api/pith-number/QP6QWOJC7JJCDMVJZPVS7CJ55L/events.json","paper":"https://pith.science/paper/QP6QWOJC"},"agent_actions":{"view_html":"https://pith.science/pith/QP6QWOJC7JJCDMVJZPVS7CJ55L","download_json":"https://pith.science/pith/QP6QWOJC7JJCDMVJZPVS7CJ55L.json","view_paper":"https://pith.science/paper/QP6QWOJC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.05644&json=true","fetch_graph":"https://pith.science/api/pith-number/QP6QWOJC7JJCDMVJZPVS7CJ55L/graph.json","fetch_events":"https://pith.science/api/pith-number/QP6QWOJC7JJCDMVJZPVS7CJ55L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QP6QWOJC7JJCDMVJZPVS7CJ55L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QP6QWOJC7JJCDMVJZPVS7CJ55L/action/storage_attestation","attest_author":"https://pith.science/pith/QP6QWOJC7JJCDMVJZPVS7CJ55L/action/author_attestation","sign_citation":"https://pith.science/pith/QP6QWOJC7JJCDMVJZPVS7CJ55L/action/citation_signature","submit_replication":"https://pith.science/pith/QP6QWOJC7JJCDMVJZPVS7CJ55L/action/replication_record"}},"created_at":"2026-07-05T09:49:47.206262+00:00","updated_at":"2026-07-05T09:49:47.206262+00:00"}