{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:3EPBZXRXEYBRKDI24UMSXX2YWE","short_pith_number":"pith:3EPBZXRX","schema_version":"1.0","canonical_sha256":"d91e1cde372603150d1ae5192bdf58b1320d95e28727fa4fcf41a7fde5bfaabd","source":{"kind":"arxiv","id":"2006.16362","version":2},"attestation_state":"computed","paper":{"title":"Multi-Head Attention: Collaborate Instead of Concatenate","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Andreas Loukas, Jean-Baptiste Cordonnier, Martin Jaggi","submitted_at":"2020-06-29T20:28:52Z","abstract_excerpt":"Attention layers are widely used in natural language processing (NLP) and are beginning to influence computer vision architectures. Training very large transformer models allowed significant improvement in both fields, but once trained, these networks show symptoms of over-parameterization. For instance, it is known that many attention heads can be pruned without impacting accuracy. This work aims to enhance current understanding on how multiple heads interact. Motivated by the observation that attention heads learn redundant key/query projections, we propose a collaborative multi-head attenti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.16362","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-06-29T20:28:52Z","cross_cats_sorted":["cs.CL","stat.ML"],"title_canon_sha256":"0508f76419d658e9105ee55c2dd9444d6d9a8dd0426f2f1d3481c5ffdacb0b5e","abstract_canon_sha256":"507b8b7c67bda8c2bd89f9076601a71697af4e27fbb775f681f73e65d9e1bd94"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:41:47.349887Z","signature_b64":"CqxYsf+0JNhTpCZx9CTs+DRcqiVC69+X6BJPQRKp/QXO2LMhsYK+Zr6b8UgYpUjwKGlPUhotBeHKo5h3AmEBDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d91e1cde372603150d1ae5192bdf58b1320d95e28727fa4fcf41a7fde5bfaabd","last_reissued_at":"2026-07-05T02:41:47.349339Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:41:47.349339Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-Head Attention: Collaborate Instead of Concatenate","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Andreas Loukas, Jean-Baptiste Cordonnier, Martin Jaggi","submitted_at":"2020-06-29T20:28:52Z","abstract_excerpt":"Attention layers are widely used in natural language processing (NLP) and are beginning to influence computer vision architectures. Training very large transformer models allowed significant improvement in both fields, but once trained, these networks show symptoms of over-parameterization. For instance, it is known that many attention heads can be pruned without impacting accuracy. This work aims to enhance current understanding on how multiple heads interact. Motivated by the observation that attention heads learn redundant key/query projections, we propose a collaborative multi-head attenti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.16362","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.16362/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.16362","created_at":"2026-07-05T02:41:47.349411+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.16362v2","created_at":"2026-07-05T02:41:47.349411+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.16362","created_at":"2026-07-05T02:41:47.349411+00:00"},{"alias_kind":"pith_short_12","alias_value":"3EPBZXRXEYBR","created_at":"2026-07-05T02:41:47.349411+00:00"},{"alias_kind":"pith_short_16","alias_value":"3EPBZXRXEYBRKDI2","created_at":"2026-07-05T02:41:47.349411+00:00"},{"alias_kind":"pith_short_8","alias_value":"3EPBZXRX","created_at":"2026-07-05T02:41:47.349411+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18967","citing_title":"EfficientRollout: System-Aware Self-Speculative Decoding for RL Rollouts","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19775","citing_title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08853","citing_title":"Architecture, Not Scale: Circuit Localization in Large Language Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05653","citing_title":"Negative Before Positive: Asymmetric Valence Processing in Large Language Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16505","citing_title":"Predicting Blastocyst Formation in IVF: Integrating DINOv2 and Attention-Based LSTM on Time-Lapse Embryo Images","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3EPBZXRXEYBRKDI24UMSXX2YWE","json":"https://pith.science/pith/3EPBZXRXEYBRKDI24UMSXX2YWE.json","graph_json":"https://pith.science/api/pith-number/3EPBZXRXEYBRKDI24UMSXX2YWE/graph.json","events_json":"https://pith.science/api/pith-number/3EPBZXRXEYBRKDI24UMSXX2YWE/events.json","paper":"https://pith.science/paper/3EPBZXRX"},"agent_actions":{"view_html":"https://pith.science/pith/3EPBZXRXEYBRKDI24UMSXX2YWE","download_json":"https://pith.science/pith/3EPBZXRXEYBRKDI24UMSXX2YWE.json","view_paper":"https://pith.science/paper/3EPBZXRX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.16362&json=true","fetch_graph":"https://pith.science/api/pith-number/3EPBZXRXEYBRKDI24UMSXX2YWE/graph.json","fetch_events":"https://pith.science/api/pith-number/3EPBZXRXEYBRKDI24UMSXX2YWE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3EPBZXRXEYBRKDI24UMSXX2YWE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3EPBZXRXEYBRKDI24UMSXX2YWE/action/storage_attestation","attest_author":"https://pith.science/pith/3EPBZXRXEYBRKDI24UMSXX2YWE/action/author_attestation","sign_citation":"https://pith.science/pith/3EPBZXRXEYBRKDI24UMSXX2YWE/action/citation_signature","submit_replication":"https://pith.science/pith/3EPBZXRXEYBRKDI24UMSXX2YWE/action/replication_record"}},"created_at":"2026-07-05T02:41:47.349411+00:00","updated_at":"2026-07-05T02:41:47.349411+00:00"}