{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:YEQZGOES4LFTZ2W7SKBCZBJ2GU","short_pith_number":"pith:YEQZGOES","schema_version":"1.0","canonical_sha256":"c121933892e2cb3ceadf92822c853a351fafea493a7a8fb628d6eb3bfb1c05ee","source":{"kind":"arxiv","id":"1905.10650","version":3},"attestation_state":"computed","paper":{"title":"Are Sixteen Heads Really Better than One?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Graham Neubig, Omer Levy, Paul Michel","submitted_at":"2019-05-25T18:27:28Z","abstract_excerpt":"Attention is a powerful and ubiquitous mechanism for allowing neural models to focus on particular salient pieces of information by taking their weighted average when making predictions. In particular, multi-headed attention is a driving force behind many recent state-of-the-art NLP models such as Transformer-based MT models and BERT. These models apply multiple attention mechanisms in parallel, with each attention \"head\" potentially focusing on different parts of the input, which makes it possible to express sophisticated functions beyond the simple weighted average. In this paper we make the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1905.10650","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-05-25T18:27:28Z","cross_cats_sorted":[],"title_canon_sha256":"7ac4e39addf7c0d2c1c701db3eb2fa77476fa701782726ec8be0f25f4b9f218e","abstract_canon_sha256":"000cb7c83ead54aed0523dd40371a049a4910dac1d4dd0a3c56410e3e39dfa8b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:16:34.482365Z","signature_b64":"1lM4TaG9FfyIZwG/vlQ9gZvxofyUkq6j54M/zSskTkjptJcYRoOYs+RiF/U1gOQQE0KlHfOEXzrZQg4Kfy0cDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c121933892e2cb3ceadf92822c853a351fafea493a7a8fb628d6eb3bfb1c05ee","last_reissued_at":"2026-07-05T00:16:34.481883Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:16:34.481883Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are Sixteen Heads Really Better than One?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Graham Neubig, Omer Levy, Paul Michel","submitted_at":"2019-05-25T18:27:28Z","abstract_excerpt":"Attention is a powerful and ubiquitous mechanism for allowing neural models to focus on particular salient pieces of information by taking their weighted average when making predictions. In particular, multi-headed attention is a driving force behind many recent state-of-the-art NLP models such as Transformer-based MT models and BERT. These models apply multiple attention mechanisms in parallel, with each attention \"head\" potentially focusing on different parts of the input, which makes it possible to express sophisticated functions beyond the simple weighted average. In this paper we make the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1905.10650","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1905.10650/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1905.10650","created_at":"2026-07-05T00:16:34.481951+00:00"},{"alias_kind":"arxiv_version","alias_value":"1905.10650v3","created_at":"2026-07-05T00:16:34.481951+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1905.10650","created_at":"2026-07-05T00:16:34.481951+00:00"},{"alias_kind":"pith_short_12","alias_value":"YEQZGOES4LFT","created_at":"2026-07-05T00:16:34.481951+00:00"},{"alias_kind":"pith_short_16","alias_value":"YEQZGOES4LFTZ2W7","created_at":"2026-07-05T00:16:34.481951+00:00"},{"alias_kind":"pith_short_8","alias_value":"YEQZGOES","created_at":"2026-07-05T00:16:34.481951+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07046","citing_title":"Voltron: Enabling Elastic Multi-Device Execution of LLM Inference for Empowered Edge Intelligence","ref_index":46,"is_internal_anchor":true},{"citing_arxiv_id":"2606.04473","citing_title":"ChessMimic: Per-Rating Transformer Models for Human Move, Clock, and Outcome Prediction in Online Blitz Chess","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"1907.00570","citing_title":"Do Transformer Attention Heads Provide Transparency in Abstractive Summarization?","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13688","citing_title":"MedCore: Boundary-Preserving Medical Core Pruning for MedSAM","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"1910.03771","citing_title":"HuggingFace's Transformers: State-of-the-art Natural Language Processing","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06694","citing_title":"AudioKV: KV Cache Eviction in Efficient Large Audio Language Models","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YEQZGOES4LFTZ2W7SKBCZBJ2GU","json":"https://pith.science/pith/YEQZGOES4LFTZ2W7SKBCZBJ2GU.json","graph_json":"https://pith.science/api/pith-number/YEQZGOES4LFTZ2W7SKBCZBJ2GU/graph.json","events_json":"https://pith.science/api/pith-number/YEQZGOES4LFTZ2W7SKBCZBJ2GU/events.json","paper":"https://pith.science/paper/YEQZGOES"},"agent_actions":{"view_html":"https://pith.science/pith/YEQZGOES4LFTZ2W7SKBCZBJ2GU","download_json":"https://pith.science/pith/YEQZGOES4LFTZ2W7SKBCZBJ2GU.json","view_paper":"https://pith.science/paper/YEQZGOES","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1905.10650&json=true","fetch_graph":"https://pith.science/api/pith-number/YEQZGOES4LFTZ2W7SKBCZBJ2GU/graph.json","fetch_events":"https://pith.science/api/pith-number/YEQZGOES4LFTZ2W7SKBCZBJ2GU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YEQZGOES4LFTZ2W7SKBCZBJ2GU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YEQZGOES4LFTZ2W7SKBCZBJ2GU/action/storage_attestation","attest_author":"https://pith.science/pith/YEQZGOES4LFTZ2W7SKBCZBJ2GU/action/author_attestation","sign_citation":"https://pith.science/pith/YEQZGOES4LFTZ2W7SKBCZBJ2GU/action/citation_signature","submit_replication":"https://pith.science/pith/YEQZGOES4LFTZ2W7SKBCZBJ2GU/action/replication_record"}},"created_at":"2026-07-05T00:16:34.481951+00:00","updated_at":"2026-07-05T00:16:34.481951+00:00"}