{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:G3JBTCHI6UD6DXMYDSKYUJVCP2","short_pith_number":"pith:G3JBTCHI","schema_version":"1.0","canonical_sha256":"36d21988e8f507e1dd981c958a26a27eb49bbf8c149c4e559fd784b79f6da004","source":{"kind":"arxiv","id":"2401.00463","version":2},"attestation_state":"computed","paper":{"title":"Analyzing Local Representations of Self-supervised Vision Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alvard Barseghyan, Ani Vanyan, Hakob Tamazyan, Hrant Khachatrian, Martin Danelljan, Vahan Huroyan","submitted_at":"2023-12-31T11:38:50Z","abstract_excerpt":"In this paper, we present a comparative analysis of various self-supervised Vision Transformers (ViTs), focusing on their local representative power. Inspired by large language models, we examine the abilities of ViTs to perform various computer vision tasks with little to no fine-tuning. We design evaluation framework to analyze the quality of local, i.e.\\ patch-level, representations in the context of few-shot semantic segmentation, instance identification, object retrieval and tracking. We discover that contrastive learning based methods like DINO produce more universal patch representation"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.00463","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-31T11:38:50Z","cross_cats_sorted":[],"title_canon_sha256":"039debf57964e3496fc29928dddb287c977921d4e4354f7f9c40cd6b5f430948","abstract_canon_sha256":"702ddce02477b0836d81d5f18628c9f034f7915de57b7584116df09f855e9417"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:59:04.696956Z","signature_b64":"DARDpWbicjaayZEmv1UIS1IXa9wfO1GEuTWkOWjJVTKp4idFCO1sm0l5Nt9deAd6sgoaTBjpkkL4kC69liPCCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"36d21988e8f507e1dd981c958a26a27eb49bbf8c149c4e559fd784b79f6da004","last_reissued_at":"2026-07-05T07:59:04.696401Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:59:04.696401Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Analyzing Local Representations of Self-supervised Vision Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alvard Barseghyan, Ani Vanyan, Hakob Tamazyan, Hrant Khachatrian, Martin Danelljan, Vahan Huroyan","submitted_at":"2023-12-31T11:38:50Z","abstract_excerpt":"In this paper, we present a comparative analysis of various self-supervised Vision Transformers (ViTs), focusing on their local representative power. Inspired by large language models, we examine the abilities of ViTs to perform various computer vision tasks with little to no fine-tuning. We design evaluation framework to analyze the quality of local, i.e.\\ patch-level, representations in the context of few-shot semantic segmentation, instance identification, object retrieval and tracking. We discover that contrastive learning based methods like DINO produce more universal patch representation"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.00463","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.00463/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.00463","created_at":"2026-07-05T07:59:04.696461+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.00463v2","created_at":"2026-07-05T07:59:04.696461+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.00463","created_at":"2026-07-05T07:59:04.696461+00:00"},{"alias_kind":"pith_short_12","alias_value":"G3JBTCHI6UD6","created_at":"2026-07-05T07:59:04.696461+00:00"},{"alias_kind":"pith_short_16","alias_value":"G3JBTCHI6UD6DXMY","created_at":"2026-07-05T07:59:04.696461+00:00"},{"alias_kind":"pith_short_8","alias_value":"G3JBTCHI","created_at":"2026-07-05T07:59:04.696461+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.10370","citing_title":"SHRUG-FM: Reliability-Aware Foundation Models for Earth Observation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06095","citing_title":"Metonymy in vision models undermines attention-based interpretability","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G3JBTCHI6UD6DXMYDSKYUJVCP2","json":"https://pith.science/pith/G3JBTCHI6UD6DXMYDSKYUJVCP2.json","graph_json":"https://pith.science/api/pith-number/G3JBTCHI6UD6DXMYDSKYUJVCP2/graph.json","events_json":"https://pith.science/api/pith-number/G3JBTCHI6UD6DXMYDSKYUJVCP2/events.json","paper":"https://pith.science/paper/G3JBTCHI"},"agent_actions":{"view_html":"https://pith.science/pith/G3JBTCHI6UD6DXMYDSKYUJVCP2","download_json":"https://pith.science/pith/G3JBTCHI6UD6DXMYDSKYUJVCP2.json","view_paper":"https://pith.science/paper/G3JBTCHI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.00463&json=true","fetch_graph":"https://pith.science/api/pith-number/G3JBTCHI6UD6DXMYDSKYUJVCP2/graph.json","fetch_events":"https://pith.science/api/pith-number/G3JBTCHI6UD6DXMYDSKYUJVCP2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G3JBTCHI6UD6DXMYDSKYUJVCP2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G3JBTCHI6UD6DXMYDSKYUJVCP2/action/storage_attestation","attest_author":"https://pith.science/pith/G3JBTCHI6UD6DXMYDSKYUJVCP2/action/author_attestation","sign_citation":"https://pith.science/pith/G3JBTCHI6UD6DXMYDSKYUJVCP2/action/citation_signature","submit_replication":"https://pith.science/pith/G3JBTCHI6UD6DXMYDSKYUJVCP2/action/replication_record"}},"created_at":"2026-07-05T07:59:04.696461+00:00","updated_at":"2026-07-05T07:59:04.696461+00:00"}