{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:CVYYHVVPYN2YNGOAOGYXLA2YNJ","short_pith_number":"pith:CVYYHVVP","schema_version":"1.0","canonical_sha256":"157183d6afc3758699c071b17583586a5c77e5a18b7b6fdf60f31b280f024d35","source":{"kind":"arxiv","id":"2209.15001","version":3},"attestation_state":"computed","paper":{"title":"Dilated Neighborhood Attention Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Ali Hassani, Humphrey Shi","submitted_at":"2022-09-29T17:57:08Z","abstract_excerpt":"Transformers are quickly becoming one of the most heavily applied deep learning architectures across modalities, domains, and tasks. In vision, on top of ongoing efforts into plain transformers, hierarchical transformers have also gained significant attention, thanks to their performance and easy integration into existing frameworks. These models typically employ localized attention mechanisms, such as the sliding-window Neighborhood Attention (NA) or Swin Transformer's Shifted Window Self Attention. While effective at reducing self attention's quadratic complexity, local attention weakens two"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.15001","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-09-29T17:57:08Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"b7c0c014d76925dfa03fb6e43cdde43e6811ee5edeba50fff6c098168be68a4e","abstract_canon_sha256":"3bafc2a3ac7ca43f88ea6a1bde18c4891017ee3ce192818a0340d124ef4ba785"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:33:18.577946Z","signature_b64":"Eqrg1Wn1G9r1jKVO1KF3L8vCg+pjZLIxo9SC4zvOtFDwLW8ivHO3Q3OJBhsoWvrlF5TaSdi/+sCiBH7ZaiIcDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"157183d6afc3758699c071b17583586a5c77e5a18b7b6fdf60f31b280f024d35","last_reissued_at":"2026-07-05T05:33:18.577444Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:33:18.577444Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dilated Neighborhood Attention Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Ali Hassani, Humphrey Shi","submitted_at":"2022-09-29T17:57:08Z","abstract_excerpt":"Transformers are quickly becoming one of the most heavily applied deep learning architectures across modalities, domains, and tasks. In vision, on top of ongoing efforts into plain transformers, hierarchical transformers have also gained significant attention, thanks to their performance and easy integration into existing frameworks. These models typically employ localized attention mechanisms, such as the sliding-window Neighborhood Attention (NA) or Swin Transformer's Shifted Window Self Attention. While effective at reducing self attention's quadratic complexity, local attention weakens two"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.15001","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.15001/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.15001","created_at":"2026-07-05T05:33:18.577505+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.15001v3","created_at":"2026-07-05T05:33:18.577505+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.15001","created_at":"2026-07-05T05:33:18.577505+00:00"},{"alias_kind":"pith_short_12","alias_value":"CVYYHVVPYN2Y","created_at":"2026-07-05T05:33:18.577505+00:00"},{"alias_kind":"pith_short_16","alias_value":"CVYYHVVPYN2YNGOA","created_at":"2026-07-05T05:33:18.577505+00:00"},{"alias_kind":"pith_short_8","alias_value":"CVYYHVVP","created_at":"2026-07-05T05:33:18.577505+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31577","citing_title":"SurGe: Improved Surface Geometry in Point Maps","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13637","citing_title":"Exploring Mutual Cross-Modal Attention for Context-Aware Human Affordance Generation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18196","citing_title":"RAT+: Train Dense, Infer Sparse -- Recurrence Augmented Attention for Dilated Inference","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05496","citing_title":"Flex Attention: A Programming Model for Generating Optimized Attention Kernels","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18196","citing_title":"RAT+: Train Dense, Infer Sparse -- Recurrence Augmented Attention for Dilated Inference","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04830","citing_title":"Concurrence of Symmetry Breaking and Nonlocality Phase Transitions in Diffusion Models","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CVYYHVVPYN2YNGOAOGYXLA2YNJ","json":"https://pith.science/pith/CVYYHVVPYN2YNGOAOGYXLA2YNJ.json","graph_json":"https://pith.science/api/pith-number/CVYYHVVPYN2YNGOAOGYXLA2YNJ/graph.json","events_json":"https://pith.science/api/pith-number/CVYYHVVPYN2YNGOAOGYXLA2YNJ/events.json","paper":"https://pith.science/paper/CVYYHVVP"},"agent_actions":{"view_html":"https://pith.science/pith/CVYYHVVPYN2YNGOAOGYXLA2YNJ","download_json":"https://pith.science/pith/CVYYHVVPYN2YNGOAOGYXLA2YNJ.json","view_paper":"https://pith.science/paper/CVYYHVVP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.15001&json=true","fetch_graph":"https://pith.science/api/pith-number/CVYYHVVPYN2YNGOAOGYXLA2YNJ/graph.json","fetch_events":"https://pith.science/api/pith-number/CVYYHVVPYN2YNGOAOGYXLA2YNJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CVYYHVVPYN2YNGOAOGYXLA2YNJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CVYYHVVPYN2YNGOAOGYXLA2YNJ/action/storage_attestation","attest_author":"https://pith.science/pith/CVYYHVVPYN2YNGOAOGYXLA2YNJ/action/author_attestation","sign_citation":"https://pith.science/pith/CVYYHVVPYN2YNGOAOGYXLA2YNJ/action/citation_signature","submit_replication":"https://pith.science/pith/CVYYHVVPYN2YNGOAOGYXLA2YNJ/action/replication_record"}},"created_at":"2026-07-05T05:33:18.577505+00:00","updated_at":"2026-07-05T05:33:18.577505+00:00"}