{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:GW4QE4IZP5NLN5JYN36N6THOZD","short_pith_number":"pith:GW4QE4IZ","schema_version":"1.0","canonical_sha256":"35b90271197f5ab6f5386efcdf4ceec8e702c534ef7678b6176ae8ef7a3d39c5","source":{"kind":"arxiv","id":"2104.03602","version":3},"attestation_state":"computed","paper":{"title":"SiT: Self-supervised vIsion Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Josef Kittler, Muhammad Awais, Sara Atito","submitted_at":"2021-04-08T08:34:04Z","abstract_excerpt":"Self-supervised learning methods are gaining increasing traction in computer vision due to their recent success in reducing the gap with supervised learning. In natural language processing (NLP) self-supervised learning and transformers are already the methods of choice. The recent literature suggests that the transformers are becoming increasingly popular also in computer vision. So far, the vision transformers have been shown to work well when pretrained either using a large scale supervised data or with some kind of co-supervision, e.g. in terms of teacher network. These supervised pretrain"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.03602","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-04-08T08:34:04Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"0ded9256c055196f060b354da9af96bb1fc29772f4752dd8501011629af886a7","abstract_canon_sha256":"15a7c2f84caf12d759b409756a99ad66a26073256209ebe74f1b38a536b5c625"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:28:41.698524Z","signature_b64":"L8YZhRnxPM8c/K28FqA4Zj51a1zA2IRutYo/f5E2/AmzrXIz6mSyj95kuB/Rm3lDdN4Jwru0JMNwzIxUWFj5DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"35b90271197f5ab6f5386efcdf4ceec8e702c534ef7678b6176ae8ef7a3d39c5","last_reissued_at":"2026-07-05T05:28:41.698104Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:28:41.698104Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SiT: Self-supervised vIsion Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Josef Kittler, Muhammad Awais, Sara Atito","submitted_at":"2021-04-08T08:34:04Z","abstract_excerpt":"Self-supervised learning methods are gaining increasing traction in computer vision due to their recent success in reducing the gap with supervised learning. In natural language processing (NLP) self-supervised learning and transformers are already the methods of choice. The recent literature suggests that the transformers are becoming increasingly popular also in computer vision. So far, the vision transformers have been shown to work well when pretrained either using a large scale supervised data or with some kind of co-supervision, e.g. in terms of teacher network. These supervised pretrain"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.03602","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.03602/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.03602","created_at":"2026-07-05T05:28:41.698157+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.03602v3","created_at":"2026-07-05T05:28:41.698157+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.03602","created_at":"2026-07-05T05:28:41.698157+00:00"},{"alias_kind":"pith_short_12","alias_value":"GW4QE4IZP5NL","created_at":"2026-07-05T05:28:41.698157+00:00"},{"alias_kind":"pith_short_16","alias_value":"GW4QE4IZP5NLN5JY","created_at":"2026-07-05T05:28:41.698157+00:00"},{"alias_kind":"pith_short_8","alias_value":"GW4QE4IZ","created_at":"2026-07-05T05:28:41.698157+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.26272","citing_title":"PRPO: Paragraph-level Policy Optimization for Vision-Language Deepfake Detection","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2111.07832","citing_title":"iBOT: Image BERT Pre-Training with Online Tokenizer","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08819","citing_title":"From pre-training to downstream performance: Does domain-specific pre-training make sense?","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GW4QE4IZP5NLN5JYN36N6THOZD","json":"https://pith.science/pith/GW4QE4IZP5NLN5JYN36N6THOZD.json","graph_json":"https://pith.science/api/pith-number/GW4QE4IZP5NLN5JYN36N6THOZD/graph.json","events_json":"https://pith.science/api/pith-number/GW4QE4IZP5NLN5JYN36N6THOZD/events.json","paper":"https://pith.science/paper/GW4QE4IZ"},"agent_actions":{"view_html":"https://pith.science/pith/GW4QE4IZP5NLN5JYN36N6THOZD","download_json":"https://pith.science/pith/GW4QE4IZP5NLN5JYN36N6THOZD.json","view_paper":"https://pith.science/paper/GW4QE4IZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.03602&json=true","fetch_graph":"https://pith.science/api/pith-number/GW4QE4IZP5NLN5JYN36N6THOZD/graph.json","fetch_events":"https://pith.science/api/pith-number/GW4QE4IZP5NLN5JYN36N6THOZD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GW4QE4IZP5NLN5JYN36N6THOZD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GW4QE4IZP5NLN5JYN36N6THOZD/action/storage_attestation","attest_author":"https://pith.science/pith/GW4QE4IZP5NLN5JYN36N6THOZD/action/author_attestation","sign_citation":"https://pith.science/pith/GW4QE4IZP5NLN5JYN36N6THOZD/action/citation_signature","submit_replication":"https://pith.science/pith/GW4QE4IZP5NLN5JYN36N6THOZD/action/replication_record"}},"created_at":"2026-07-05T05:28:41.698157+00:00","updated_at":"2026-07-05T05:28:41.698157+00:00"}