{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KAYZUY2A42F4UKYFIOL7CBZTJQ","short_pith_number":"pith:KAYZUY2A","schema_version":"1.0","canonical_sha256":"50319a6340e68bca2b054397f107334c067a5d9669b0c209b5b9ed62a67f229d","source":{"kind":"arxiv","id":"2410.02746","version":2},"attestation_state":"computed","paper":{"title":"Contrastive Localized Language-Image Pre-Training","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Bowen Zhang, Haotian Zhang, Hong-You Chen, Keen You, Marcin Eichner, Meng Cao, Xinze Wang, Yinfei Yang, Zhe Gan, Zhengfeng Lai","submitted_at":"2024-10-03T17:56:09Z","abstract_excerpt":"Contrastive Language-Image Pre-training (CLIP) has been a celebrated method for training vision encoders to generate image/text representations facilitating various applications. Recently, CLIP has been widely adopted as the vision backbone of multimodal large language models (MLLMs) to connect image inputs for language interactions. The success of CLIP as a vision-language foundation model relies on aligning web-crawled noisy text annotations at image levels. Nevertheless, such criteria may become insufficient for downstream tasks in need of fine-grained vision representations, especially whe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.02746","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-03T17:56:09Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"84f6f4166f63610a545fdb4b5a736619c2846fa4e75d8b7c6c281181d5425d1f","abstract_canon_sha256":"92086bf0cf86f2b1937a071e9fd123012473340ce50ebe7ba811b35fa3efece5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:16:38.323420Z","signature_b64":"sJaGEn7aFPD3Wq4xVGUdyfX8565OdAaF6qED6gerSKqTJ1N1+/xC8UkJmm1WY5mHP99qjfKVB6QAf/6O1+flAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"50319a6340e68bca2b054397f107334c067a5d9669b0c209b5b9ed62a67f229d","last_reissued_at":"2026-07-05T10:16:38.322823Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:16:38.322823Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Contrastive Localized Language-Image Pre-Training","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Bowen Zhang, Haotian Zhang, Hong-You Chen, Keen You, Marcin Eichner, Meng Cao, Xinze Wang, Yinfei Yang, Zhe Gan, Zhengfeng Lai","submitted_at":"2024-10-03T17:56:09Z","abstract_excerpt":"Contrastive Language-Image Pre-training (CLIP) has been a celebrated method for training vision encoders to generate image/text representations facilitating various applications. Recently, CLIP has been widely adopted as the vision backbone of multimodal large language models (MLLMs) to connect image inputs for language interactions. The success of CLIP as a vision-language foundation model relies on aligning web-crawled noisy text annotations at image levels. Nevertheless, such criteria may become insufficient for downstream tasks in need of fine-grained vision representations, especially whe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.02746","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.02746/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.02746","created_at":"2026-07-05T10:16:38.322886+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.02746v2","created_at":"2026-07-05T10:16:38.322886+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.02746","created_at":"2026-07-05T10:16:38.322886+00:00"},{"alias_kind":"pith_short_12","alias_value":"KAYZUY2A42F4","created_at":"2026-07-05T10:16:38.322886+00:00"},{"alias_kind":"pith_short_16","alias_value":"KAYZUY2A42F4UKYF","created_at":"2026-07-05T10:16:38.322886+00:00"},{"alias_kind":"pith_short_8","alias_value":"KAYZUY2A","created_at":"2026-07-05T10:16:38.322886+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19489","citing_title":"Concept Flow Models: Anchoring Concept-Based Reasoning with Hierarchical Bottlenecks","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01825","citing_title":"ROGLE: Robust Global-Local Alignment with Automated Region Supervision for Text-Based Person Search","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01825","citing_title":"ROGLE: Robust Global-Local Alignment with Automated Region Supervision for Text-Based Person Search","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29776","citing_title":"Improving CLIP Adaptation by Breaking Tail Alignment for Source-Free Cross-Domain Few-Shot Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09552","citing_title":"MCERF: Advancing Multimodal LLM Evaluation of Engineering Documentation with Enhanced Retrieval","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KAYZUY2A42F4UKYFIOL7CBZTJQ","json":"https://pith.science/pith/KAYZUY2A42F4UKYFIOL7CBZTJQ.json","graph_json":"https://pith.science/api/pith-number/KAYZUY2A42F4UKYFIOL7CBZTJQ/graph.json","events_json":"https://pith.science/api/pith-number/KAYZUY2A42F4UKYFIOL7CBZTJQ/events.json","paper":"https://pith.science/paper/KAYZUY2A"},"agent_actions":{"view_html":"https://pith.science/pith/KAYZUY2A42F4UKYFIOL7CBZTJQ","download_json":"https://pith.science/pith/KAYZUY2A42F4UKYFIOL7CBZTJQ.json","view_paper":"https://pith.science/paper/KAYZUY2A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.02746&json=true","fetch_graph":"https://pith.science/api/pith-number/KAYZUY2A42F4UKYFIOL7CBZTJQ/graph.json","fetch_events":"https://pith.science/api/pith-number/KAYZUY2A42F4UKYFIOL7CBZTJQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KAYZUY2A42F4UKYFIOL7CBZTJQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KAYZUY2A42F4UKYFIOL7CBZTJQ/action/storage_attestation","attest_author":"https://pith.science/pith/KAYZUY2A42F4UKYFIOL7CBZTJQ/action/author_attestation","sign_citation":"https://pith.science/pith/KAYZUY2A42F4UKYFIOL7CBZTJQ/action/citation_signature","submit_replication":"https://pith.science/pith/KAYZUY2A42F4UKYFIOL7CBZTJQ/action/replication_record"}},"created_at":"2026-07-05T10:16:38.322886+00:00","updated_at":"2026-07-05T10:16:38.322886+00:00"}