{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UYXDYTROB4FEYBFADPFQYNAQ7W","short_pith_number":"pith:UYXDYTRO","schema_version":"1.0","canonical_sha256":"a62e3c4e2e0f0a4c04a01bcb0c3410fdb23decb3cef25b960331143263d83592","source":{"kind":"arxiv","id":"2410.19836","version":2},"attestation_state":"computed","paper":{"title":"Upsampling DINOv2 features for unsupervised vision tasks and weakly supervised materials segmentation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.mtrl-sci","eess.IV"],"primary_cat":"cs.CV","authors_text":"Antonis Vamvakeros, Ronan Docherty, Samuel J. Cooper","submitted_at":"2024-10-20T13:01:53Z","abstract_excerpt":"The features of self-supervised vision transformers (ViTs) contain strong semantic and positional information relevant to downstream tasks like object localization and segmentation. Recent works combine these features with traditional methods like clustering, graph partitioning or region correlations to achieve impressive baselines without finetuning or training additional networks. We leverage upsampled features from ViT networks (e.g DINOv2) in two workflows: in a clustering based approach for object localization and segmentation, and paired with standard classifiers in weakly supervised mat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.19836","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-20T13:01:53Z","cross_cats_sorted":["cond-mat.mtrl-sci","eess.IV"],"title_canon_sha256":"6600fee4010b5c1a2b63526f087af7382a52e2ad8701b54eea8771ef00c1e847","abstract_canon_sha256":"c9b4cecceb389b3b4cbc2b55cbc6e39b2adce76eb34334f4c96ebd4583fd8299"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:04.928386Z","signature_b64":"6KVDtI8F1QxhcHpbrFMF0K/Ywu2wraHtmDXxjXaa50Wct9QLQtKZZfTq0ynapASuSG9cPI98tG9MCpz+IYd8CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a62e3c4e2e0f0a4c04a01bcb0c3410fdb23decb3cef25b960331143263d83592","last_reissued_at":"2026-07-05T11:49:04.927952Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:04.927952Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Upsampling DINOv2 features for unsupervised vision tasks and weakly supervised materials segmentation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.mtrl-sci","eess.IV"],"primary_cat":"cs.CV","authors_text":"Antonis Vamvakeros, Ronan Docherty, Samuel J. Cooper","submitted_at":"2024-10-20T13:01:53Z","abstract_excerpt":"The features of self-supervised vision transformers (ViTs) contain strong semantic and positional information relevant to downstream tasks like object localization and segmentation. Recent works combine these features with traditional methods like clustering, graph partitioning or region correlations to achieve impressive baselines without finetuning or training additional networks. We leverage upsampled features from ViT networks (e.g DINOv2) in two workflows: in a clustering based approach for object localization and segmentation, and paired with standard classifiers in weakly supervised mat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.19836","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.19836/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.19836","created_at":"2026-07-05T11:49:04.928008+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.19836v2","created_at":"2026-07-05T11:49:04.928008+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.19836","created_at":"2026-07-05T11:49:04.928008+00:00"},{"alias_kind":"pith_short_12","alias_value":"UYXDYTROB4FE","created_at":"2026-07-05T11:49:04.928008+00:00"},{"alias_kind":"pith_short_16","alias_value":"UYXDYTROB4FEYBFA","created_at":"2026-07-05T11:49:04.928008+00:00"},{"alias_kind":"pith_short_8","alias_value":"UYXDYTRO","created_at":"2026-07-05T11:49:04.928008+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.06912","citing_title":"PANC: Prior-Aware Normalized Cut via Anchor-Augmented Token Graphs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14582","citing_title":"MapSR: Prompt-Driven Land Cover Map Super-Resolution via Vision Foundation Models","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UYXDYTROB4FEYBFADPFQYNAQ7W","json":"https://pith.science/pith/UYXDYTROB4FEYBFADPFQYNAQ7W.json","graph_json":"https://pith.science/api/pith-number/UYXDYTROB4FEYBFADPFQYNAQ7W/graph.json","events_json":"https://pith.science/api/pith-number/UYXDYTROB4FEYBFADPFQYNAQ7W/events.json","paper":"https://pith.science/paper/UYXDYTRO"},"agent_actions":{"view_html":"https://pith.science/pith/UYXDYTROB4FEYBFADPFQYNAQ7W","download_json":"https://pith.science/pith/UYXDYTROB4FEYBFADPFQYNAQ7W.json","view_paper":"https://pith.science/paper/UYXDYTRO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.19836&json=true","fetch_graph":"https://pith.science/api/pith-number/UYXDYTROB4FEYBFADPFQYNAQ7W/graph.json","fetch_events":"https://pith.science/api/pith-number/UYXDYTROB4FEYBFADPFQYNAQ7W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UYXDYTROB4FEYBFADPFQYNAQ7W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UYXDYTROB4FEYBFADPFQYNAQ7W/action/storage_attestation","attest_author":"https://pith.science/pith/UYXDYTROB4FEYBFADPFQYNAQ7W/action/author_attestation","sign_citation":"https://pith.science/pith/UYXDYTROB4FEYBFADPFQYNAQ7W/action/citation_signature","submit_replication":"https://pith.science/pith/UYXDYTROB4FEYBFADPFQYNAQ7W/action/replication_record"}},"created_at":"2026-07-05T11:49:04.928008+00:00","updated_at":"2026-07-05T11:49:04.928008+00:00"}