{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:SM2HJMCHN63HL44RAL3CHNGVCD","short_pith_number":"pith:SM2HJMCH","schema_version":"1.0","canonical_sha256":"933474b0476fb675f39102f623b4d510fde437211f42a95e588693356cd1da80","source":{"kind":"arxiv","id":"2105.10497","version":3},"attestation_state":"computed","paper":{"title":"Intriguing Properties of Vision Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Fahad Shahbaz Khan, Kanchana Ranasinghe, Ming-Hsuan Yang, Munawar Hayat, Muzammal Naseer, Salman Khan","submitted_at":"2021-05-21T17:59:18Z","abstract_excerpt":"Vision transformers (ViT) have demonstrated impressive performance across various machine vision problems. These models are based on multi-head self-attention mechanisms that can flexibly attend to a sequence of image patches to encode contextual cues. An important question is how such flexibility in attending image-wide context conditioned on a given patch can facilitate handling nuisances in natural images e.g., severe occlusions, domain shifts, spatial permutations, adversarial and natural perturbations. We systematically study this question via an extensive set of experiments encompassing "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.10497","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-05-21T17:59:18Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"ab8f98b988547e99584ceba4108965d56ea9d91d8bff11e3ec05d0176e1ca4df","abstract_canon_sha256":"4381b07a22d1a474f6d8cba09b1cdb77be45cac69daa3ae1ca41592db9e46213"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:35:10.084503Z","signature_b64":"UO/yzUHPmGYuYHLZ6dPmF5zjXmMEg6dJZlJePnFHtT1geGaAeLubpniT8UafDQIPhUAqFjcNMa4VTUTWGu+ECA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"933474b0476fb675f39102f623b4d510fde437211f42a95e588693356cd1da80","last_reissued_at":"2026-07-05T03:35:10.084057Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:35:10.084057Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Intriguing Properties of Vision Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Fahad Shahbaz Khan, Kanchana Ranasinghe, Ming-Hsuan Yang, Munawar Hayat, Muzammal Naseer, Salman Khan","submitted_at":"2021-05-21T17:59:18Z","abstract_excerpt":"Vision transformers (ViT) have demonstrated impressive performance across various machine vision problems. These models are based on multi-head self-attention mechanisms that can flexibly attend to a sequence of image patches to encode contextual cues. An important question is how such flexibility in attending image-wide context conditioned on a given patch can facilitate handling nuisances in natural images e.g., severe occlusions, domain shifts, spatial permutations, adversarial and natural perturbations. We systematically study this question via an extensive set of experiments encompassing "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.10497","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.10497/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.10497","created_at":"2026-07-05T03:35:10.084112+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.10497v3","created_at":"2026-07-05T03:35:10.084112+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.10497","created_at":"2026-07-05T03:35:10.084112+00:00"},{"alias_kind":"pith_short_12","alias_value":"SM2HJMCHN63H","created_at":"2026-07-05T03:35:10.084112+00:00"},{"alias_kind":"pith_short_16","alias_value":"SM2HJMCHN63HL44R","created_at":"2026-07-05T03:35:10.084112+00:00"},{"alias_kind":"pith_short_8","alias_value":"SM2HJMCH","created_at":"2026-07-05T03:35:10.084112+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18510","citing_title":"Architectural Bias in Face Presentation Attack Detection: A Comparative Study of Vision Transformers and Convolutional Neural Networks","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03795","citing_title":"Beyond Compression: Quantifying Spectral Accessibility in Vision Representations","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2111.07832","citing_title":"iBOT: Image BERT Pre-Training with Online Tokenizer","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SM2HJMCHN63HL44RAL3CHNGVCD","json":"https://pith.science/pith/SM2HJMCHN63HL44RAL3CHNGVCD.json","graph_json":"https://pith.science/api/pith-number/SM2HJMCHN63HL44RAL3CHNGVCD/graph.json","events_json":"https://pith.science/api/pith-number/SM2HJMCHN63HL44RAL3CHNGVCD/events.json","paper":"https://pith.science/paper/SM2HJMCH"},"agent_actions":{"view_html":"https://pith.science/pith/SM2HJMCHN63HL44RAL3CHNGVCD","download_json":"https://pith.science/pith/SM2HJMCHN63HL44RAL3CHNGVCD.json","view_paper":"https://pith.science/paper/SM2HJMCH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.10497&json=true","fetch_graph":"https://pith.science/api/pith-number/SM2HJMCHN63HL44RAL3CHNGVCD/graph.json","fetch_events":"https://pith.science/api/pith-number/SM2HJMCHN63HL44RAL3CHNGVCD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SM2HJMCHN63HL44RAL3CHNGVCD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SM2HJMCHN63HL44RAL3CHNGVCD/action/storage_attestation","attest_author":"https://pith.science/pith/SM2HJMCHN63HL44RAL3CHNGVCD/action/author_attestation","sign_citation":"https://pith.science/pith/SM2HJMCHN63HL44RAL3CHNGVCD/action/citation_signature","submit_replication":"https://pith.science/pith/SM2HJMCHN63HL44RAL3CHNGVCD/action/replication_record"}},"created_at":"2026-07-05T03:35:10.084112+00:00","updated_at":"2026-07-05T03:35:10.084112+00:00"}