{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:H3WLULBSDR57CQ55TL7O43SRHS","short_pith_number":"pith:H3WLULBS","schema_version":"1.0","canonical_sha256":"3eecba2c321c7bf143bd9afeee6e513caca50e0330d4000e4cb0be1bcba1de57","source":{"kind":"arxiv","id":"2006.15595","version":4},"attestation_state":"computed","paper":{"title":"Rethinking Positional Encoding in Language Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Di He, Guolin Ke, Tie-Yan Liu","submitted_at":"2020-06-28T13:11:02Z","abstract_excerpt":"In this work, we investigate the positional encoding methods used in language pre-training (e.g., BERT) and identify several problems in the existing formulations. First, we show that in the absolute positional encoding, the addition operation applied on positional embeddings and word embeddings brings mixed correlations between the two heterogeneous information resources. It may bring unnecessary randomness in the attention and further limit the expressiveness of the model. Second, we question whether treating the position of the symbol \\texttt{[CLS]} the same as other words is a reasonable d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.15595","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-06-28T13:11:02Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"c36e38abc177f56ddc749330d8fe3453a45068880eccba37f5a01a1fb712f83f","abstract_canon_sha256":"bd684b8360c67e87aa429338b63d252adad185dde6e2d135daa80487aa9176bd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:22:36.789935Z","signature_b64":"ei268NttUOX2NqMFG8Y3r8D2BTVW5/jYNwRjWqY4OO5DJEvBIMFUCb1Jkb5xN63W30Ofy3/GVbA7q9kf38PuCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3eecba2c321c7bf143bd9afeee6e513caca50e0330d4000e4cb0be1bcba1de57","last_reissued_at":"2026-07-05T02:22:36.789535Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:22:36.789535Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Rethinking Positional Encoding in Language Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Di He, Guolin Ke, Tie-Yan Liu","submitted_at":"2020-06-28T13:11:02Z","abstract_excerpt":"In this work, we investigate the positional encoding methods used in language pre-training (e.g., BERT) and identify several problems in the existing formulations. First, we show that in the absolute positional encoding, the addition operation applied on positional embeddings and word embeddings brings mixed correlations between the two heterogeneous information resources. It may bring unnecessary randomness in the attention and further limit the expressiveness of the model. Second, we question whether treating the position of the symbol \\texttt{[CLS]} the same as other words is a reasonable d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.15595","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.15595/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.15595","created_at":"2026-07-05T02:22:36.789596+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.15595v4","created_at":"2026-07-05T02:22:36.789596+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.15595","created_at":"2026-07-05T02:22:36.789596+00:00"},{"alias_kind":"pith_short_12","alias_value":"H3WLULBSDR57","created_at":"2026-07-05T02:22:36.789596+00:00"},{"alias_kind":"pith_short_16","alias_value":"H3WLULBSDR57CQ55","created_at":"2026-07-05T02:22:36.789596+00:00"},{"alias_kind":"pith_short_8","alias_value":"H3WLULBS","created_at":"2026-07-05T02:22:36.789596+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31585","citing_title":"DPPE: Rethinking Camera-Based Positional Encoding for Scaling Multi-View Transformers","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30022","citing_title":"Give it Space! Explicit Disentangling of Positional and Semantic Representations in Encoders","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2502.12370","citing_title":"Positional Encoding in Transformer-Based Time Series Models: A Survey","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2504.14386","citing_title":"LOOPE: Learnable Optimal Patch Order in Positional Embeddings for Vision Transformers","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2509.12635","citing_title":"Positional Encoding via Token-Aware Phase Attention","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2509.14640","citing_title":"DyWPE: Signal-Aware Dynamic Wavelet Positional Encoding for Time Series Transformers","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2512.07805","citing_title":"Group Representational Position Encoding","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2402.17762","citing_title":"Massive Activations in Large Language Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2402.06196","citing_title":"Large Language Models: A Survey","ref_index":129,"is_internal_anchor":false},{"citing_arxiv_id":"2104.09864","citing_title":"RoFormer: Enhanced Transformer with Rotary Position Embedding","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H3WLULBSDR57CQ55TL7O43SRHS","json":"https://pith.science/pith/H3WLULBSDR57CQ55TL7O43SRHS.json","graph_json":"https://pith.science/api/pith-number/H3WLULBSDR57CQ55TL7O43SRHS/graph.json","events_json":"https://pith.science/api/pith-number/H3WLULBSDR57CQ55TL7O43SRHS/events.json","paper":"https://pith.science/paper/H3WLULBS"},"agent_actions":{"view_html":"https://pith.science/pith/H3WLULBSDR57CQ55TL7O43SRHS","download_json":"https://pith.science/pith/H3WLULBSDR57CQ55TL7O43SRHS.json","view_paper":"https://pith.science/paper/H3WLULBS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.15595&json=true","fetch_graph":"https://pith.science/api/pith-number/H3WLULBSDR57CQ55TL7O43SRHS/graph.json","fetch_events":"https://pith.science/api/pith-number/H3WLULBSDR57CQ55TL7O43SRHS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H3WLULBSDR57CQ55TL7O43SRHS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H3WLULBSDR57CQ55TL7O43SRHS/action/storage_attestation","attest_author":"https://pith.science/pith/H3WLULBSDR57CQ55TL7O43SRHS/action/author_attestation","sign_citation":"https://pith.science/pith/H3WLULBSDR57CQ55TL7O43SRHS/action/citation_signature","submit_replication":"https://pith.science/pith/H3WLULBSDR57CQ55TL7O43SRHS/action/replication_record"}},"created_at":"2026-07-05T02:22:36.789596+00:00","updated_at":"2026-07-05T02:22:36.789596+00:00"}