{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TWH4U3MSTN5HJMUCUL3VYUVPKD","short_pith_number":"pith:TWH4U3MS","schema_version":"1.0","canonical_sha256":"9d8fca6d929b7a74b282a2f75c52af50f8cfbd87a4eb5d10e35252a6b032589c","source":{"kind":"arxiv","id":"2405.13335","version":2},"attestation_state":"computed","paper":{"title":"Vision Transformer with Sparse Scan Prior","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Huaibo Huang, Qihang Fan, Yuguang Zhang","submitted_at":"2024-05-22T04:34:36Z","abstract_excerpt":"In recent years, Transformers have achieved remarkable progress in computer vision tasks. However, their global modeling often comes with substantial computational overhead, in stark contrast to the human eye's efficient information processing. Inspired by the human eye's sparse scanning mechanism, we propose a \\textbf{S}parse \\textbf{S}can \\textbf{S}elf-\\textbf{A}ttention mechanism ($\\rm{S}^3\\rm{A}$). This mechanism predefines a series of Anchors of Interest for each token and employs local attention to efficiently model the spatial information around these anchors, avoiding redundant global "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.13335","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-22T04:34:36Z","cross_cats_sorted":[],"title_canon_sha256":"48aefb7067a44458020eff7853bf86608ca2f5ef2c34128a1d66ca0cc19f0960","abstract_canon_sha256":"daecc08a0b97d86ac78b9faaf3898be60e4321d6cd721a9b610f269d3f7ed8b1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:07:53.852590Z","signature_b64":"JRXeaBNmdIc5IIEKlMA/AalrLmyTJZZOdvrAdyIISmM0X0WALEgL0N+QG1r9NF3NybNB42kg5JrCbpNG3vOIAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d8fca6d929b7a74b282a2f75c52af50f8cfbd87a4eb5d10e35252a6b032589c","last_reissued_at":"2026-07-05T12:07:53.852032Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:07:53.852032Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vision Transformer with Sparse Scan Prior","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Huaibo Huang, Qihang Fan, Yuguang Zhang","submitted_at":"2024-05-22T04:34:36Z","abstract_excerpt":"In recent years, Transformers have achieved remarkable progress in computer vision tasks. However, their global modeling often comes with substantial computational overhead, in stark contrast to the human eye's efficient information processing. Inspired by the human eye's sparse scanning mechanism, we propose a \\textbf{S}parse \\textbf{S}can \\textbf{S}elf-\\textbf{A}ttention mechanism ($\\rm{S}^3\\rm{A}$). This mechanism predefines a series of Anchors of Interest for each token and employs local attention to efficiently model the spatial information around these anchors, avoiding redundant global "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.13335","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.13335/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.13335","created_at":"2026-07-05T12:07:53.852090+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.13335v2","created_at":"2026-07-05T12:07:53.852090+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.13335","created_at":"2026-07-05T12:07:53.852090+00:00"},{"alias_kind":"pith_short_12","alias_value":"TWH4U3MSTN5H","created_at":"2026-07-05T12:07:53.852090+00:00"},{"alias_kind":"pith_short_16","alias_value":"TWH4U3MSTN5HJMUC","created_at":"2026-07-05T12:07:53.852090+00:00"},{"alias_kind":"pith_short_8","alias_value":"TWH4U3MS","created_at":"2026-07-05T12:07:53.852090+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.20836","citing_title":"HAD: Hybrid Architecture Distillation Outperforms Teacher in Genomic Sequence Modeling","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TWH4U3MSTN5HJMUCUL3VYUVPKD","json":"https://pith.science/pith/TWH4U3MSTN5HJMUCUL3VYUVPKD.json","graph_json":"https://pith.science/api/pith-number/TWH4U3MSTN5HJMUCUL3VYUVPKD/graph.json","events_json":"https://pith.science/api/pith-number/TWH4U3MSTN5HJMUCUL3VYUVPKD/events.json","paper":"https://pith.science/paper/TWH4U3MS"},"agent_actions":{"view_html":"https://pith.science/pith/TWH4U3MSTN5HJMUCUL3VYUVPKD","download_json":"https://pith.science/pith/TWH4U3MSTN5HJMUCUL3VYUVPKD.json","view_paper":"https://pith.science/paper/TWH4U3MS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.13335&json=true","fetch_graph":"https://pith.science/api/pith-number/TWH4U3MSTN5HJMUCUL3VYUVPKD/graph.json","fetch_events":"https://pith.science/api/pith-number/TWH4U3MSTN5HJMUCUL3VYUVPKD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TWH4U3MSTN5HJMUCUL3VYUVPKD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TWH4U3MSTN5HJMUCUL3VYUVPKD/action/storage_attestation","attest_author":"https://pith.science/pith/TWH4U3MSTN5HJMUCUL3VYUVPKD/action/author_attestation","sign_citation":"https://pith.science/pith/TWH4U3MSTN5HJMUCUL3VYUVPKD/action/citation_signature","submit_replication":"https://pith.science/pith/TWH4U3MSTN5HJMUCUL3VYUVPKD/action/replication_record"}},"created_at":"2026-07-05T12:07:53.852090+00:00","updated_at":"2026-07-05T12:07:53.852090+00:00"}