{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BBJD5Y4GPS5JGJM27V6HEMYZYV","short_pith_number":"pith:BBJD5Y4G","schema_version":"1.0","canonical_sha256":"08523ee3867cba93259afd7c723319c578813eaf354ac6576498caab460d1df8","source":{"kind":"arxiv","id":"2407.20171","version":4},"attestation_state":"computed","paper":{"title":"Diffusion Feedback Helps CLIP See Better","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fan Zhang, Jing Liu, Quan Sun, Wenxuan Wang, Xinlong Wang, Yepeng Tang","submitted_at":"2024-07-29T17:00:09Z","abstract_excerpt":"Contrastive Language-Image Pre-training (CLIP), which excels at abstracting open-world representations across domains and modalities, has become a foundation for a variety of vision and multimodal tasks. However, recent studies reveal that CLIP has severe visual shortcomings, such as which can hardly distinguish orientation, quantity, color, structure, etc. These visual shortcomings also limit the perception capabilities of multimodal large language models (MLLMs) built on CLIP. The main reason could be that the image-text pairs used to train CLIP are inherently biased, due to the lack of the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.20171","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-29T17:00:09Z","cross_cats_sorted":[],"title_canon_sha256":"459eee5833165ea95430c0f8745bf90f53bc6df53720806c37ec8b82a79859ce","abstract_canon_sha256":"abc53ae494faaa4f0bca55cb62b6bfebcae91f76d92d44c60ce901b69ac1e2ff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:58:46.367990Z","signature_b64":"1EY5mEx4O7Li4+0V/aFTP42YgQt125XniYzaKmm4xWCxLLi/IKRWcmDkpPbM8TNqq739QDsi3JBWwTaVIGaEDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"08523ee3867cba93259afd7c723319c578813eaf354ac6576498caab460d1df8","last_reissued_at":"2026-07-05T08:58:46.367541Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:58:46.367541Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Diffusion Feedback Helps CLIP See Better","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fan Zhang, Jing Liu, Quan Sun, Wenxuan Wang, Xinlong Wang, Yepeng Tang","submitted_at":"2024-07-29T17:00:09Z","abstract_excerpt":"Contrastive Language-Image Pre-training (CLIP), which excels at abstracting open-world representations across domains and modalities, has become a foundation for a variety of vision and multimodal tasks. However, recent studies reveal that CLIP has severe visual shortcomings, such as which can hardly distinguish orientation, quantity, color, structure, etc. These visual shortcomings also limit the perception capabilities of multimodal large language models (MLLMs) built on CLIP. The main reason could be that the image-text pairs used to train CLIP are inherently biased, due to the lack of the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.20171","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.20171/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.20171","created_at":"2026-07-05T08:58:46.367589+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.20171v4","created_at":"2026-07-05T08:58:46.367589+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.20171","created_at":"2026-07-05T08:58:46.367589+00:00"},{"alias_kind":"pith_short_12","alias_value":"BBJD5Y4GPS5J","created_at":"2026-07-05T08:58:46.367589+00:00"},{"alias_kind":"pith_short_16","alias_value":"BBJD5Y4GPS5JGJM2","created_at":"2026-07-05T08:58:46.367589+00:00"},{"alias_kind":"pith_short_8","alias_value":"BBJD5Y4G","created_at":"2026-07-05T08:58:46.367589+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18714","citing_title":"Semantic Generative Tuning for Unified Multimodal Models","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2507.23372","citing_title":"UniEmo: Unifying Emotional Understanding and Generation with Learnable Expert Queries","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16384","citing_title":"Mutual Enhancement Between Global Tokens and Patch Tokens: From Theory to Practice","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18714","citing_title":"Semantic Generative Tuning for Unified Multimodal Models","ref_index":63,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BBJD5Y4GPS5JGJM27V6HEMYZYV","json":"https://pith.science/pith/BBJD5Y4GPS5JGJM27V6HEMYZYV.json","graph_json":"https://pith.science/api/pith-number/BBJD5Y4GPS5JGJM27V6HEMYZYV/graph.json","events_json":"https://pith.science/api/pith-number/BBJD5Y4GPS5JGJM27V6HEMYZYV/events.json","paper":"https://pith.science/paper/BBJD5Y4G"},"agent_actions":{"view_html":"https://pith.science/pith/BBJD5Y4GPS5JGJM27V6HEMYZYV","download_json":"https://pith.science/pith/BBJD5Y4GPS5JGJM27V6HEMYZYV.json","view_paper":"https://pith.science/paper/BBJD5Y4G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.20171&json=true","fetch_graph":"https://pith.science/api/pith-number/BBJD5Y4GPS5JGJM27V6HEMYZYV/graph.json","fetch_events":"https://pith.science/api/pith-number/BBJD5Y4GPS5JGJM27V6HEMYZYV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BBJD5Y4GPS5JGJM27V6HEMYZYV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BBJD5Y4GPS5JGJM27V6HEMYZYV/action/storage_attestation","attest_author":"https://pith.science/pith/BBJD5Y4GPS5JGJM27V6HEMYZYV/action/author_attestation","sign_citation":"https://pith.science/pith/BBJD5Y4GPS5JGJM27V6HEMYZYV/action/citation_signature","submit_replication":"https://pith.science/pith/BBJD5Y4GPS5JGJM27V6HEMYZYV/action/replication_record"}},"created_at":"2026-07-05T08:58:46.367589+00:00","updated_at":"2026-07-05T08:58:46.367589+00:00"}