{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:3LXIAZ7SKB7DIF6O35BIS3HMPO","short_pith_number":"pith:3LXIAZ7S","schema_version":"1.0","canonical_sha256":"daee8067f2507e3417cedf42896cec7b9e0fc68c88370634fa900704fe9932e1","source":{"kind":"arxiv","id":"2106.01950","version":1},"attestation_state":"computed","paper":{"title":"The Case for Translation-Invariant Self-Attention in Transformer-Based Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Gustav Eje Henter, Ulme Wennberg","submitted_at":"2021-06-03T15:56:26Z","abstract_excerpt":"Mechanisms for encoding positional information are central for transformer-based language models. In this paper, we analyze the position embeddings of existing language models, finding strong evidence of translation invariance, both for the embeddings themselves and for their effect on self-attention. The degree of translation invariance increases during training and correlates positively with model performance. Our findings lead us to propose translation-invariant self-attention (TISA), which accounts for the relative position between tokens in an interpretable fashion without needing convent"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.01950","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-06-03T15:56:26Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"9f3a25b4f68842588857ebd9860e92315dcbac7ee794adc677b39b682843f985","abstract_canon_sha256":"123a56d6eaaf5402a8a8848a90d822d25a7d3153b3b06d0aa2b31b5005ab5029"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:46:06.004803Z","signature_b64":"Db2jPUHkgTnbTtRzrxNFHvo6jRQNN1jKodUoM1x1WxQYIfoOZZ1uvWo3RKXpQmQvaMBEGd4+NduzwDeoQtTCCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"daee8067f2507e3417cedf42896cec7b9e0fc68c88370634fa900704fe9932e1","last_reissued_at":"2026-07-05T02:46:06.004342Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:46:06.004342Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Case for Translation-Invariant Self-Attention in Transformer-Based Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Gustav Eje Henter, Ulme Wennberg","submitted_at":"2021-06-03T15:56:26Z","abstract_excerpt":"Mechanisms for encoding positional information are central for transformer-based language models. In this paper, we analyze the position embeddings of existing language models, finding strong evidence of translation invariance, both for the embeddings themselves and for their effect on self-attention. The degree of translation invariance increases during training and correlates positively with model performance. Our findings lead us to propose translation-invariant self-attention (TISA), which accounts for the relative position between tokens in an interpretable fashion without needing convent"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.01950","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.01950/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.01950","created_at":"2026-07-05T02:46:06.004402+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.01950v1","created_at":"2026-07-05T02:46:06.004402+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.01950","created_at":"2026-07-05T02:46:06.004402+00:00"},{"alias_kind":"pith_short_12","alias_value":"3LXIAZ7SKB7D","created_at":"2026-07-05T02:46:06.004402+00:00"},{"alias_kind":"pith_short_16","alias_value":"3LXIAZ7SKB7DIF6O","created_at":"2026-07-05T02:46:06.004402+00:00"},{"alias_kind":"pith_short_8","alias_value":"3LXIAZ7S","created_at":"2026-07-05T02:46:06.004402+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.07805","citing_title":"Group Representational Position Encoding","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3LXIAZ7SKB7DIF6O35BIS3HMPO","json":"https://pith.science/pith/3LXIAZ7SKB7DIF6O35BIS3HMPO.json","graph_json":"https://pith.science/api/pith-number/3LXIAZ7SKB7DIF6O35BIS3HMPO/graph.json","events_json":"https://pith.science/api/pith-number/3LXIAZ7SKB7DIF6O35BIS3HMPO/events.json","paper":"https://pith.science/paper/3LXIAZ7S"},"agent_actions":{"view_html":"https://pith.science/pith/3LXIAZ7SKB7DIF6O35BIS3HMPO","download_json":"https://pith.science/pith/3LXIAZ7SKB7DIF6O35BIS3HMPO.json","view_paper":"https://pith.science/paper/3LXIAZ7S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.01950&json=true","fetch_graph":"https://pith.science/api/pith-number/3LXIAZ7SKB7DIF6O35BIS3HMPO/graph.json","fetch_events":"https://pith.science/api/pith-number/3LXIAZ7SKB7DIF6O35BIS3HMPO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3LXIAZ7SKB7DIF6O35BIS3HMPO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3LXIAZ7SKB7DIF6O35BIS3HMPO/action/storage_attestation","attest_author":"https://pith.science/pith/3LXIAZ7SKB7DIF6O35BIS3HMPO/action/author_attestation","sign_citation":"https://pith.science/pith/3LXIAZ7SKB7DIF6O35BIS3HMPO/action/citation_signature","submit_replication":"https://pith.science/pith/3LXIAZ7SKB7DIF6O35BIS3HMPO/action/replication_record"}},"created_at":"2026-07-05T02:46:06.004402+00:00","updated_at":"2026-07-05T02:46:06.004402+00:00"}