{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LNECD6SG2PHXO4QGHWIOYZIEPQ","short_pith_number":"pith:LNECD6SG","schema_version":"1.0","canonical_sha256":"5b4821fa46d3cf7772063d90ec65047c0499c5bcbad27eb0c1d4ef4193673c91","source":{"kind":"arxiv","id":"2505.09466","version":1},"attestation_state":"computed","paper":{"title":"A 2D Semantic-Aware Position Encoding for Vision Transformers","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Biao Yang, Chuan Li, Feng Shi, Huishuai Bao, Jiaxu Feng, Kun Zhou, Muqi Huang, Shiyang Zhou, Sijia Peng, Xi Chen, Yuhui Zhang, Yun Xiong","submitted_at":"2025-05-14T15:17:34Z","abstract_excerpt":"Vision transformers have demonstrated significant advantages in computer vision tasks due to their ability to capture long-range dependencies and contextual relationships through self-attention. However, existing position encoding techniques, which are largely borrowed from natural language processing, fail to effectively capture semantic-aware positional relationships between image patches. Traditional approaches like absolute position encoding and relative position encoding primarily focus on 1D linear position relationship, often neglecting the semantic similarity between distant yet contex"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.09466","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-14T15:17:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a5c78328a0c52d0c1132237968d03eec2e949ebb8e7ebe4e122e41def230a6a1","abstract_canon_sha256":"455106271d4be0cf04f15a79334f4e6d9800ab9b6fcc2c96977babd5481400ae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:03:08.937489Z","signature_b64":"nBG3/pnSQn4gANoZVD3z/lov4s17SfyCF8dcVPZG5GYbYS5OqiYwbIRmWorpPQQ5Zn/JKde/yRw+qYe1VduWCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5b4821fa46d3cf7772063d90ec65047c0499c5bcbad27eb0c1d4ef4193673c91","last_reissued_at":"2026-07-05T11:03:08.936998Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:03:08.936998Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A 2D Semantic-Aware Position Encoding for Vision Transformers","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Biao Yang, Chuan Li, Feng Shi, Huishuai Bao, Jiaxu Feng, Kun Zhou, Muqi Huang, Shiyang Zhou, Sijia Peng, Xi Chen, Yuhui Zhang, Yun Xiong","submitted_at":"2025-05-14T15:17:34Z","abstract_excerpt":"Vision transformers have demonstrated significant advantages in computer vision tasks due to their ability to capture long-range dependencies and contextual relationships through self-attention. However, existing position encoding techniques, which are largely borrowed from natural language processing, fail to effectively capture semantic-aware positional relationships between image patches. Traditional approaches like absolute position encoding and relative position encoding primarily focus on 1D linear position relationship, often neglecting the semantic similarity between distant yet contex"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.09466","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.09466/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.09466","created_at":"2026-07-05T11:03:08.937056+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.09466v1","created_at":"2026-07-05T11:03:08.937056+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.09466","created_at":"2026-07-05T11:03:08.937056+00:00"},{"alias_kind":"pith_short_12","alias_value":"LNECD6SG2PHX","created_at":"2026-07-05T11:03:08.937056+00:00"},{"alias_kind":"pith_short_16","alias_value":"LNECD6SG2PHXO4QG","created_at":"2026-07-05T11:03:08.937056+00:00"},{"alias_kind":"pith_short_8","alias_value":"LNECD6SG","created_at":"2026-07-05T11:03:08.937056+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00580","citing_title":"Active Spatial Guidance: Eliminating Injected Positional Mechanisms in Vision Transformers","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LNECD6SG2PHXO4QGHWIOYZIEPQ","json":"https://pith.science/pith/LNECD6SG2PHXO4QGHWIOYZIEPQ.json","graph_json":"https://pith.science/api/pith-number/LNECD6SG2PHXO4QGHWIOYZIEPQ/graph.json","events_json":"https://pith.science/api/pith-number/LNECD6SG2PHXO4QGHWIOYZIEPQ/events.json","paper":"https://pith.science/paper/LNECD6SG"},"agent_actions":{"view_html":"https://pith.science/pith/LNECD6SG2PHXO4QGHWIOYZIEPQ","download_json":"https://pith.science/pith/LNECD6SG2PHXO4QGHWIOYZIEPQ.json","view_paper":"https://pith.science/paper/LNECD6SG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.09466&json=true","fetch_graph":"https://pith.science/api/pith-number/LNECD6SG2PHXO4QGHWIOYZIEPQ/graph.json","fetch_events":"https://pith.science/api/pith-number/LNECD6SG2PHXO4QGHWIOYZIEPQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LNECD6SG2PHXO4QGHWIOYZIEPQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LNECD6SG2PHXO4QGHWIOYZIEPQ/action/storage_attestation","attest_author":"https://pith.science/pith/LNECD6SG2PHXO4QGHWIOYZIEPQ/action/author_attestation","sign_citation":"https://pith.science/pith/LNECD6SG2PHXO4QGHWIOYZIEPQ/action/citation_signature","submit_replication":"https://pith.science/pith/LNECD6SG2PHXO4QGHWIOYZIEPQ/action/replication_record"}},"created_at":"2026-07-05T11:03:08.937056+00:00","updated_at":"2026-07-05T11:03:08.937056+00:00"}