{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:P6FWJFHEXHJFWKAE2KJ25LKNIL","short_pith_number":"pith:P6FWJFHE","schema_version":"1.0","canonical_sha256":"7f8b6494e4b9d25b2804d293aead4d42c6f5cd5cca4858d0fd4d9881b0660c3c","source":{"kind":"arxiv","id":"2106.09650","version":1},"attestation_state":"computed","paper":{"title":"Multi-head or Single-head? An Empirical Comparison for Transformer Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jialu Liu, Jiawei Han, Liyuan Liu","submitted_at":"2021-06-17T16:53:22Z","abstract_excerpt":"Multi-head attention plays a crucial role in the recent success of Transformer models, which leads to consistent performance improvements over conventional attention in various applications. The popular belief is that this effectiveness stems from the ability of jointly attending multiple positions. In this paper, we first demonstrate that jointly attending multiple positions is not a unique feature of multi-head attention, as multi-layer single-head attention also attends multiple positions and is more effective. Then, we suggest the main advantage of the multi-head attention is the training "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.09650","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-06-17T16:53:22Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"61bd4869abd1ce766f2e519c6da9e809cd7532684fd91e81f2a6c6f6d85a3229","abstract_canon_sha256":"cc1e5315c5ea89c0110a6320e6343dd36b55420ee7bce0d2922c84596a134d85"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:50:18.280508Z","signature_b64":"kIdFRHBlNzYKwk92p+AO9QFqVsujyd6eDQ0PwsrfmDHGGQdRdJwNlm6GepxmSKE5rnvGSTVeYOabWi2I1YxEAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7f8b6494e4b9d25b2804d293aead4d42c6f5cd5cca4858d0fd4d9881b0660c3c","last_reissued_at":"2026-07-05T02:50:18.279954Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:50:18.279954Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-head or Single-head? An Empirical Comparison for Transformer Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jialu Liu, Jiawei Han, Liyuan Liu","submitted_at":"2021-06-17T16:53:22Z","abstract_excerpt":"Multi-head attention plays a crucial role in the recent success of Transformer models, which leads to consistent performance improvements over conventional attention in various applications. The popular belief is that this effectiveness stems from the ability of jointly attending multiple positions. In this paper, we first demonstrate that jointly attending multiple positions is not a unique feature of multi-head attention, as multi-layer single-head attention also attends multiple positions and is more effective. Then, we suggest the main advantage of the multi-head attention is the training "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.09650","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.09650/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.09650","created_at":"2026-07-05T02:50:18.280028+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.09650v1","created_at":"2026-07-05T02:50:18.280028+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.09650","created_at":"2026-07-05T02:50:18.280028+00:00"},{"alias_kind":"pith_short_12","alias_value":"P6FWJFHEXHJF","created_at":"2026-07-05T02:50:18.280028+00:00"},{"alias_kind":"pith_short_16","alias_value":"P6FWJFHEXHJFWKAE","created_at":"2026-07-05T02:50:18.280028+00:00"},{"alias_kind":"pith_short_8","alias_value":"P6FWJFHE","created_at":"2026-07-05T02:50:18.280028+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01432","citing_title":"Leaf Spectral Reflectance Prediction Using Multi-Head Attention Neural Networks","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25315","citing_title":"SaliencyDecor: Enhancing Neural Network Interpretability through Feature Decorrelation","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P6FWJFHEXHJFWKAE2KJ25LKNIL","json":"https://pith.science/pith/P6FWJFHEXHJFWKAE2KJ25LKNIL.json","graph_json":"https://pith.science/api/pith-number/P6FWJFHEXHJFWKAE2KJ25LKNIL/graph.json","events_json":"https://pith.science/api/pith-number/P6FWJFHEXHJFWKAE2KJ25LKNIL/events.json","paper":"https://pith.science/paper/P6FWJFHE"},"agent_actions":{"view_html":"https://pith.science/pith/P6FWJFHEXHJFWKAE2KJ25LKNIL","download_json":"https://pith.science/pith/P6FWJFHEXHJFWKAE2KJ25LKNIL.json","view_paper":"https://pith.science/paper/P6FWJFHE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.09650&json=true","fetch_graph":"https://pith.science/api/pith-number/P6FWJFHEXHJFWKAE2KJ25LKNIL/graph.json","fetch_events":"https://pith.science/api/pith-number/P6FWJFHEXHJFWKAE2KJ25LKNIL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P6FWJFHEXHJFWKAE2KJ25LKNIL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P6FWJFHEXHJFWKAE2KJ25LKNIL/action/storage_attestation","attest_author":"https://pith.science/pith/P6FWJFHEXHJFWKAE2KJ25LKNIL/action/author_attestation","sign_citation":"https://pith.science/pith/P6FWJFHEXHJFWKAE2KJ25LKNIL/action/citation_signature","submit_replication":"https://pith.science/pith/P6FWJFHEXHJFWKAE2KJ25LKNIL/action/replication_record"}},"created_at":"2026-07-05T02:50:18.280028+00:00","updated_at":"2026-07-05T02:50:18.280028+00:00"}