{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EK4QNLNQ6V3SFLU2T44GPQ5P6D","short_pith_number":"pith:EK4QNLNQ","schema_version":"1.0","canonical_sha256":"22b906adb0f57722ae9a9f3867c3aff0d804d43ccf51c364b855eaacfc840b82","source":{"kind":"arxiv","id":"2412.03513","version":2},"attestation_state":"computed","paper":{"title":"Enhancing CLIP Conceptual Embedding through Knowledge Distillation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.AI","authors_text":"Kuei-Chun Kao","submitted_at":"2024-12-04T17:56:49Z","abstract_excerpt":"Recently, CLIP has become an important model for aligning images and text in multi-modal contexts. However, researchers have identified limitations in the ability of CLIP's text and image encoders to extract detailed knowledge from pairs of captions and images. In response, this paper presents Knowledge-CLIP, an innovative approach designed to improve CLIP's performance by integrating a new knowledge distillation (KD) method based on Llama 2. Our approach focuses on three key objectives: Text Embedding Distillation, Concept Learning, and Contrastive Learning. First, Text Embedding Distillation"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.03513","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2024-12-04T17:56:49Z","cross_cats_sorted":["cs.CL","cs.CV","cs.LG"],"title_canon_sha256":"f3f59b97bd6ed04d37f8c6016ec462ac05aa75ee323fedba2b81f1e1f2c1cbb1","abstract_canon_sha256":"b22e00391178495497bb92248aac98d12798e759ee2bb135cfd42c6373640866"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:59.903676Z","signature_b64":"3WakUTC5X4CS0VbbiYC81IurEN3+2/Uy4cgStyfZDQHIXlMj0ImTcj1u7kqSztQHkoCZC17V4OqGthYpYcZ7DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"22b906adb0f57722ae9a9f3867c3aff0d804d43ccf51c364b855eaacfc840b82","last_reissued_at":"2026-07-05T09:45:59.903160Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:59.903160Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enhancing CLIP Conceptual Embedding through Knowledge Distillation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.AI","authors_text":"Kuei-Chun Kao","submitted_at":"2024-12-04T17:56:49Z","abstract_excerpt":"Recently, CLIP has become an important model for aligning images and text in multi-modal contexts. However, researchers have identified limitations in the ability of CLIP's text and image encoders to extract detailed knowledge from pairs of captions and images. In response, this paper presents Knowledge-CLIP, an innovative approach designed to improve CLIP's performance by integrating a new knowledge distillation (KD) method based on Llama 2. Our approach focuses on three key objectives: Text Embedding Distillation, Concept Learning, and Contrastive Learning. First, Text Embedding Distillation"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.03513","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.03513/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.03513","created_at":"2026-07-05T09:45:59.903227+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.03513v2","created_at":"2026-07-05T09:45:59.903227+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.03513","created_at":"2026-07-05T09:45:59.903227+00:00"},{"alias_kind":"pith_short_12","alias_value":"EK4QNLNQ6V3S","created_at":"2026-07-05T09:45:59.903227+00:00"},{"alias_kind":"pith_short_16","alias_value":"EK4QNLNQ6V3SFLU2","created_at":"2026-07-05T09:45:59.903227+00:00"},{"alias_kind":"pith_short_8","alias_value":"EK4QNLNQ","created_at":"2026-07-05T09:45:59.903227+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EK4QNLNQ6V3SFLU2T44GPQ5P6D","json":"https://pith.science/pith/EK4QNLNQ6V3SFLU2T44GPQ5P6D.json","graph_json":"https://pith.science/api/pith-number/EK4QNLNQ6V3SFLU2T44GPQ5P6D/graph.json","events_json":"https://pith.science/api/pith-number/EK4QNLNQ6V3SFLU2T44GPQ5P6D/events.json","paper":"https://pith.science/paper/EK4QNLNQ"},"agent_actions":{"view_html":"https://pith.science/pith/EK4QNLNQ6V3SFLU2T44GPQ5P6D","download_json":"https://pith.science/pith/EK4QNLNQ6V3SFLU2T44GPQ5P6D.json","view_paper":"https://pith.science/paper/EK4QNLNQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.03513&json=true","fetch_graph":"https://pith.science/api/pith-number/EK4QNLNQ6V3SFLU2T44GPQ5P6D/graph.json","fetch_events":"https://pith.science/api/pith-number/EK4QNLNQ6V3SFLU2T44GPQ5P6D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EK4QNLNQ6V3SFLU2T44GPQ5P6D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EK4QNLNQ6V3SFLU2T44GPQ5P6D/action/storage_attestation","attest_author":"https://pith.science/pith/EK4QNLNQ6V3SFLU2T44GPQ5P6D/action/author_attestation","sign_citation":"https://pith.science/pith/EK4QNLNQ6V3SFLU2T44GPQ5P6D/action/citation_signature","submit_replication":"https://pith.science/pith/EK4QNLNQ6V3SFLU2T44GPQ5P6D/action/replication_record"}},"created_at":"2026-07-05T09:45:59.903227+00:00","updated_at":"2026-07-05T09:45:59.903227+00:00"}