{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:HJ47EALBXUFUQSWQ5ABSD4D7JX","short_pith_number":"pith:HJ47EALB","schema_version":"1.0","canonical_sha256":"3a79f20161bd0b484ad0e80321f07f4ddf09959369d961e1b550c6fec7a5dd90","source":{"kind":"arxiv","id":"2204.03645","version":1},"attestation_state":"computed","paper":{"title":"DaViT: Dual Attention Vision Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bin Xiao, Jingdong Wang, Lu Yuan, Mingyu Ding, Noel Codella, Ping Luo","submitted_at":"2022-04-07T17:59:32Z","abstract_excerpt":"In this work, we introduce Dual Attention Vision Transformers (DaViT), a simple yet effective vision transformer architecture that is able to capture global context while maintaining computational efficiency. We propose approaching the problem from an orthogonal angle: exploiting self-attention mechanisms with both \"spatial tokens\" and \"channel tokens\". With spatial tokens, the spatial dimension defines the token scope, and the channel dimension defines the token feature dimension. With channel tokens, we have the inverse: the channel dimension defines the token scope, and the spatial dimensio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.03645","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-04-07T17:59:32Z","cross_cats_sorted":[],"title_canon_sha256":"dbd8b0221032472753175047d0a0f987f593069fe18927cf9f1e99c55ae84de9","abstract_canon_sha256":"45118b447eb0fa7cd6a8f43f70b9d283c61992853a3604c2f451fd60cea3d0d8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:12:32.566117Z","signature_b64":"Jz6RJxi1MioWfjfIFbKQPfD2+a1vs29b98OcUkdn/Tag6zPMwbw34xKiibWHMDWmP1AXZ+tVdAeJQKgYWBV7CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3a79f20161bd0b484ad0e80321f07f4ddf09959369d961e1b550c6fec7a5dd90","last_reissued_at":"2026-07-05T04:12:32.565753Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:12:32.565753Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DaViT: Dual Attention Vision Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bin Xiao, Jingdong Wang, Lu Yuan, Mingyu Ding, Noel Codella, Ping Luo","submitted_at":"2022-04-07T17:59:32Z","abstract_excerpt":"In this work, we introduce Dual Attention Vision Transformers (DaViT), a simple yet effective vision transformer architecture that is able to capture global context while maintaining computational efficiency. We propose approaching the problem from an orthogonal angle: exploiting self-attention mechanisms with both \"spatial tokens\" and \"channel tokens\". With spatial tokens, the spatial dimension defines the token scope, and the channel dimension defines the token feature dimension. With channel tokens, we have the inverse: the channel dimension defines the token scope, and the spatial dimensio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.03645","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.03645/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.03645","created_at":"2026-07-05T04:12:32.565816+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.03645v1","created_at":"2026-07-05T04:12:32.565816+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.03645","created_at":"2026-07-05T04:12:32.565816+00:00"},{"alias_kind":"pith_short_12","alias_value":"HJ47EALBXUFU","created_at":"2026-07-05T04:12:32.565816+00:00"},{"alias_kind":"pith_short_16","alias_value":"HJ47EALBXUFUQSWQ","created_at":"2026-07-05T04:12:32.565816+00:00"},{"alias_kind":"pith_short_8","alias_value":"HJ47EALB","created_at":"2026-07-05T04:12:32.565816+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.18683","citing_title":"SIM-Net: A Multimodal Fusion Network Using Inferred 3D Object Shape Point Clouds from RGB Images for 2D Classification","ref_index":55,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HJ47EALBXUFUQSWQ5ABSD4D7JX","json":"https://pith.science/pith/HJ47EALBXUFUQSWQ5ABSD4D7JX.json","graph_json":"https://pith.science/api/pith-number/HJ47EALBXUFUQSWQ5ABSD4D7JX/graph.json","events_json":"https://pith.science/api/pith-number/HJ47EALBXUFUQSWQ5ABSD4D7JX/events.json","paper":"https://pith.science/paper/HJ47EALB"},"agent_actions":{"view_html":"https://pith.science/pith/HJ47EALBXUFUQSWQ5ABSD4D7JX","download_json":"https://pith.science/pith/HJ47EALBXUFUQSWQ5ABSD4D7JX.json","view_paper":"https://pith.science/paper/HJ47EALB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.03645&json=true","fetch_graph":"https://pith.science/api/pith-number/HJ47EALBXUFUQSWQ5ABSD4D7JX/graph.json","fetch_events":"https://pith.science/api/pith-number/HJ47EALBXUFUQSWQ5ABSD4D7JX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HJ47EALBXUFUQSWQ5ABSD4D7JX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HJ47EALBXUFUQSWQ5ABSD4D7JX/action/storage_attestation","attest_author":"https://pith.science/pith/HJ47EALBXUFUQSWQ5ABSD4D7JX/action/author_attestation","sign_citation":"https://pith.science/pith/HJ47EALBXUFUQSWQ5ABSD4D7JX/action/citation_signature","submit_replication":"https://pith.science/pith/HJ47EALBXUFUQSWQ5ABSD4D7JX/action/replication_record"}},"created_at":"2026-07-05T04:12:32.565816+00:00","updated_at":"2026-07-05T04:12:32.565816+00:00"}