{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OYG2SSMOALXTLXJ4VBXLDGQW3S","short_pith_number":"pith:OYG2SSMO","schema_version":"1.0","canonical_sha256":"760da9498e02ef35dd3ca86eb19a16dc9e343fc9d908ad7b57a76047aaf2bc68","source":{"kind":"arxiv","id":"2404.16637","version":1},"attestation_state":"computed","paper":{"title":"Zero-Shot Distillation for Image Encoders: How to Make Effective Use of Synthetic Data","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jan Hendrik Metzen, Matthias Hein, Niclas Popp","submitted_at":"2024-04-25T14:24:41Z","abstract_excerpt":"Multi-modal foundation models such as CLIP have showcased impressive zero-shot capabilities. However, their applicability in resource-constrained environments is limited due to their large number of parameters and high inference time. While existing approaches have scaled down the entire CLIP architecture, we focus on training smaller variants of the image encoder, which suffices for efficient zero-shot classification. The use of synthetic data has shown promise in distilling representations from larger teachers, resulting in strong few-shot and linear probe performance. However, we find that "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.16637","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CV","submitted_at":"2024-04-25T14:24:41Z","cross_cats_sorted":[],"title_canon_sha256":"b3e1f1efadad9188a3b3cdd40fe2f01abb505d8a2331723dbfa65310cf97ff5d","abstract_canon_sha256":"55d16b8fec6678124afb2666d3a975a55e2a2508e03269812b4a03a4a06565bd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:12:10.619190Z","signature_b64":"VgPI3rTFYBCX4gELAyA1yHLUNtTWX5VuB6DhKY5Q/KK7upxmB61J6ERj2sgBJYDwRoQl+8nigqtLMswQS60qAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"760da9498e02ef35dd3ca86eb19a16dc9e343fc9d908ad7b57a76047aaf2bc68","last_reissued_at":"2026-07-05T08:12:10.618637Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:12:10.618637Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Zero-Shot Distillation for Image Encoders: How to Make Effective Use of Synthetic Data","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jan Hendrik Metzen, Matthias Hein, Niclas Popp","submitted_at":"2024-04-25T14:24:41Z","abstract_excerpt":"Multi-modal foundation models such as CLIP have showcased impressive zero-shot capabilities. However, their applicability in resource-constrained environments is limited due to their large number of parameters and high inference time. While existing approaches have scaled down the entire CLIP architecture, we focus on training smaller variants of the image encoder, which suffices for efficient zero-shot classification. The use of synthetic data has shown promise in distilling representations from larger teachers, resulting in strong few-shot and linear probe performance. However, we find that "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.16637","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.16637/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.16637","created_at":"2026-07-05T08:12:10.618700+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.16637v1","created_at":"2026-07-05T08:12:10.618700+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.16637","created_at":"2026-07-05T08:12:10.618700+00:00"},{"alias_kind":"pith_short_12","alias_value":"OYG2SSMOALXT","created_at":"2026-07-05T08:12:10.618700+00:00"},{"alias_kind":"pith_short_16","alias_value":"OYG2SSMOALXTLXJ4","created_at":"2026-07-05T08:12:10.618700+00:00"},{"alias_kind":"pith_short_8","alias_value":"OYG2SSMO","created_at":"2026-07-05T08:12:10.618700+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.24380","citing_title":"Structural Pruning of Large Vision Language Models: A Comprehensive Study on Pruning Dynamics, Recovery, and Data Efficiency","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OYG2SSMOALXTLXJ4VBXLDGQW3S","json":"https://pith.science/pith/OYG2SSMOALXTLXJ4VBXLDGQW3S.json","graph_json":"https://pith.science/api/pith-number/OYG2SSMOALXTLXJ4VBXLDGQW3S/graph.json","events_json":"https://pith.science/api/pith-number/OYG2SSMOALXTLXJ4VBXLDGQW3S/events.json","paper":"https://pith.science/paper/OYG2SSMO"},"agent_actions":{"view_html":"https://pith.science/pith/OYG2SSMOALXTLXJ4VBXLDGQW3S","download_json":"https://pith.science/pith/OYG2SSMOALXTLXJ4VBXLDGQW3S.json","view_paper":"https://pith.science/paper/OYG2SSMO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.16637&json=true","fetch_graph":"https://pith.science/api/pith-number/OYG2SSMOALXTLXJ4VBXLDGQW3S/graph.json","fetch_events":"https://pith.science/api/pith-number/OYG2SSMOALXTLXJ4VBXLDGQW3S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OYG2SSMOALXTLXJ4VBXLDGQW3S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OYG2SSMOALXTLXJ4VBXLDGQW3S/action/storage_attestation","attest_author":"https://pith.science/pith/OYG2SSMOALXTLXJ4VBXLDGQW3S/action/author_attestation","sign_citation":"https://pith.science/pith/OYG2SSMOALXTLXJ4VBXLDGQW3S/action/citation_signature","submit_replication":"https://pith.science/pith/OYG2SSMOALXTLXJ4VBXLDGQW3S/action/replication_record"}},"created_at":"2026-07-05T08:12:10.618700+00:00","updated_at":"2026-07-05T08:12:10.618700+00:00"}