{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:COHEZDHPNJR2ROJI5TBBJBZKIK","short_pith_number":"pith:COHEZDHP","schema_version":"1.0","canonical_sha256":"138e4c8cef6a63a8b928ecc214872a42aa4d61c8ea7244eb1353d9d9ac9ff29c","source":{"kind":"arxiv","id":"2009.13658","version":1},"attestation_state":"computed","paper":{"title":"Improve Transformer Models with Better Relative Position Embeddings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bing Xiang, Davis Liang, Peng Xu, Zhiheng Huang","submitted_at":"2020-09-28T22:18:58Z","abstract_excerpt":"Transformer architectures rely on explicit position encodings in order to preserve a notion of word order. In this paper, we argue that existing work does not fully utilize position information. For example, the initial proposal of a sinusoid embedding is fixed and not learnable. In this paper, we first review absolute position embeddings and existing methods for relative position embeddings. We then propose new techniques that encourage increased interaction between query, key and relative position embeddings in the self-attention mechanism. Our most promising approach is a generalization of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2009.13658","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-09-28T22:18:58Z","cross_cats_sorted":[],"title_canon_sha256":"2ac59a4be1171cdffde47c1adc939fb46f7441920a9bc64ee2929bac796c135b","abstract_canon_sha256":"ef34463205de5b92c2f787bca782f9f905c00905c2598b09a28df1ae096e5607"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:38:43.705886Z","signature_b64":"xrYhymRmgxEvZLo8HEmSZHYjqHw3r1D9DzuaMNEJVWOnhwRXUQ6yKmhzmty+ujgCn2oNnjY/qmvp6NoJ+VlGBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"138e4c8cef6a63a8b928ecc214872a42aa4d61c8ea7244eb1353d9d9ac9ff29c","last_reissued_at":"2026-07-05T01:38:43.705389Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:38:43.705389Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improve Transformer Models with Better Relative Position Embeddings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bing Xiang, Davis Liang, Peng Xu, Zhiheng Huang","submitted_at":"2020-09-28T22:18:58Z","abstract_excerpt":"Transformer architectures rely on explicit position encodings in order to preserve a notion of word order. In this paper, we argue that existing work does not fully utilize position information. For example, the initial proposal of a sinusoid embedding is fixed and not learnable. In this paper, we first review absolute position embeddings and existing methods for relative position embeddings. We then propose new techniques that encourage increased interaction between query, key and relative position embeddings in the self-attention mechanism. Our most promising approach is a generalization of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2009.13658","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2009.13658/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2009.13658","created_at":"2026-07-05T01:38:43.705456+00:00"},{"alias_kind":"arxiv_version","alias_value":"2009.13658v1","created_at":"2026-07-05T01:38:43.705456+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2009.13658","created_at":"2026-07-05T01:38:43.705456+00:00"},{"alias_kind":"pith_short_12","alias_value":"COHEZDHPNJR2","created_at":"2026-07-05T01:38:43.705456+00:00"},{"alias_kind":"pith_short_16","alias_value":"COHEZDHPNJR2ROJI","created_at":"2026-07-05T01:38:43.705456+00:00"},{"alias_kind":"pith_short_8","alias_value":"COHEZDHP","created_at":"2026-07-05T01:38:43.705456+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27705","citing_title":"Mitigating Position Bias in Transformers via Layer-Specific Positional Embedding Scaling","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2502.12370","citing_title":"Positional Encoding in Transformer-Based Time Series Models: A Survey","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/COHEZDHPNJR2ROJI5TBBJBZKIK","json":"https://pith.science/pith/COHEZDHPNJR2ROJI5TBBJBZKIK.json","graph_json":"https://pith.science/api/pith-number/COHEZDHPNJR2ROJI5TBBJBZKIK/graph.json","events_json":"https://pith.science/api/pith-number/COHEZDHPNJR2ROJI5TBBJBZKIK/events.json","paper":"https://pith.science/paper/COHEZDHP"},"agent_actions":{"view_html":"https://pith.science/pith/COHEZDHPNJR2ROJI5TBBJBZKIK","download_json":"https://pith.science/pith/COHEZDHPNJR2ROJI5TBBJBZKIK.json","view_paper":"https://pith.science/paper/COHEZDHP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2009.13658&json=true","fetch_graph":"https://pith.science/api/pith-number/COHEZDHPNJR2ROJI5TBBJBZKIK/graph.json","fetch_events":"https://pith.science/api/pith-number/COHEZDHPNJR2ROJI5TBBJBZKIK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/COHEZDHPNJR2ROJI5TBBJBZKIK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/COHEZDHPNJR2ROJI5TBBJBZKIK/action/storage_attestation","attest_author":"https://pith.science/pith/COHEZDHPNJR2ROJI5TBBJBZKIK/action/author_attestation","sign_citation":"https://pith.science/pith/COHEZDHPNJR2ROJI5TBBJBZKIK/action/citation_signature","submit_replication":"https://pith.science/pith/COHEZDHPNJR2ROJI5TBBJBZKIK/action/replication_record"}},"created_at":"2026-07-05T01:38:43.705456+00:00","updated_at":"2026-07-05T01:38:43.705456+00:00"}