{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:U6DHRDMH5ZOVSFEGBMUK5UUQWL","short_pith_number":"pith:U6DHRDMH","schema_version":"1.0","canonical_sha256":"a786788d87ee5d5914860b28aed290b2e210266b3d8bd2d4444804119baacdf8","source":{"kind":"arxiv","id":"2104.08718","version":3},"attestation_state":"computed","paper":{"title":"CLIPScore: A Reference-free Evaluation Metric for Image Captioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"CLIP embeddings can score how well a generated caption matches its image without any human reference captions and match human judgments better than metrics that require them.","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Ari Holtzman, Jack Hessel, Maxwell Forbes, Ronan Le Bras, Yejin Choi","submitted_at":"2021-04-18T05:00:29Z","abstract_excerpt":"Image captioning has conventionally relied on reference-based automatic evaluations, where machine captions are compared against captions written by humans. This is in contrast to the reference-free manner in which humans assess caption quality.\n  In this paper, we report the surprising empirical finding that CLIP (Radford et al., 2021), a cross-modal model pretrained on 400M image+caption pairs from the web, can be used for robust automatic evaluation of image captioning without the need for references. Experiments spanning several corpora demonstrate that our new reference-free metric, CLIPS"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":true,"formal_links_present":true},"canonical_record":{"source":{"id":"2104.08718","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-04-18T05:00:29Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"5a05653470e8c550a92d4fe65e0f4ece7b773022769b49c34d79b20ef59fa93f","abstract_canon_sha256":"1f674c0ec38e27587c85a02d6f5bcd24bb575172db7ea4d02f8b4121441f531c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:08:03.170201Z","signature_b64":"amDaslhue2+xBf5GfEeZlfeg7ua5nJkUN1q+zWv3DCdC0NCzMxAxlSZiEPYXlCfBcQKcqocJu0NjBvKLEQOlBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a786788d87ee5d5914860b28aed290b2e210266b3d8bd2d4444804119baacdf8","last_reissued_at":"2026-07-05T04:08:03.169682Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:08:03.169682Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CLIPScore: A Reference-free Evaluation Metric for Image Captioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"CLIP embeddings can score how well a generated caption matches its image without any human reference captions and match human judgments better than metrics that require them.","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Ari Holtzman, Jack Hessel, Maxwell Forbes, Ronan Le Bras, Yejin Choi","submitted_at":"2021-04-18T05:00:29Z","abstract_excerpt":"Image captioning has conventionally relied on reference-based automatic evaluations, where machine captions are compared against captions written by humans. This is in contrast to the reference-free manner in which humans assess caption quality.\n  In this paper, we report the surprising empirical finding that CLIP (Radford et al., 2021), a cross-modal model pretrained on 400M image+caption pairs from the web, can be used for robust automatic evaluation of image captioning without the need for references. Experiments spanning several corpora demonstrate that our new reference-free metric, CLIPS"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"CLIPScore achieves the highest correlation with human judgements, outperforming existing reference-based metrics like CIDEr and SPICE.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That CLIP's representations pretrained on web data provide a robust, general signal of caption quality that transfers across domains without needing task-specific adaptation or references.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"CLIPScore uses a web-pretrained CLIP model to evaluate image captions without references and achieves higher human correlation than CIDEr or SPICE.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"CLIP embeddings can score how well a generated caption matches its image without any human reference captions and match human judgments better than metrics that require them.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"25c8b47e90ba3b8006754fbdb5d4d6377cf057ff3a96bb095bfb2f4cce773d21"},"source":{"id":"2104.08718","kind":"arxiv","version":3},"verdict":{"id":"e9258c23-c1fd-418b-84cf-9cb4b2aedcaf","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-12T22:19:02.195685Z","strongest_claim":"CLIPScore achieves the highest correlation with human judgements, outperforming existing reference-based metrics like CIDEr and SPICE.","one_line_summary":"CLIPScore uses a web-pretrained CLIP model to evaluate image captions without references and achieves higher human correlation than CIDEr or SPICE.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That CLIP's representations pretrained on web data provide a robust, general signal of caption quality that transfers across domains without needing task-specific adaptation or references.","pith_extraction_headline":"CLIP embeddings can score how well a generated caption matches its image without any human reference captions and match human judgments better than metrics that require them."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.08718/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":70,"sample":[{"doi":"","year":2015,"title":"From Images to Sentences through Scene Description Graphs using Commonsense Reasoning and Knowledge","work_id":"dd4c908a-9217-463d-852e-1034998769f6","ref_index":1,"cited_arxiv_id":"1511.03292","is_internal_anchor":false},{"doi":"","year":2021,"title":"Evaluating clip: towards characterization of broader capabilities and downstream implications.arXiv preprint arXiv:2108.02818","work_id":"7161697e-26f5-4f20-b765-eaa45dfbbc04","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2016,"title":"Peter Anderson, Basura Fernando, Mark Johnson, and Stephen Gould. 2016. Spice: Semantic propositional image caption evaluation. In ECCV. Springer","work_id":"0ea2ce8e-367d-446e-a1e5-00dc7c7a43a8","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2019,"title":"Mikel Artetxe and Holger Schwenk. 2019. Massively multilingual sentence embeddings for zero-shot cross-lingual transfer and beyond. TACL, 7:597--610","work_id":"04b2b247-a772-418c-b7dc-0ec31f152fff","ref_index":4,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2005,"title":"Satanjeev Banerjee and Alon Lavie. 2005. METEOR: an automatic metric for mt evaluation with improved correlation with human judgments. In ACL workshop on Evaluation Measures for MT and Summarization","work_id":"41abb198-7a92-44c0-80ee-e7010f8bb6e1","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":70,"snapshot_sha256":"8cb28f3cb0c8c4cbcfa45256280a5950952845475368a3c2946e74df34d751e5","internal_anchors":1},"formal_canon":{"evidence_count":2,"snapshot_sha256":"e70a7474b2ab98fc43167f394859d7b551a6f4479c663187747cd01fe74d0fad"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.08718","created_at":"2026-07-05T04:08:03.169753+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.08718v3","created_at":"2026-07-05T04:08:03.169753+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.08718","created_at":"2026-07-05T04:08:03.169753+00:00"},{"alias_kind":"pith_short_12","alias_value":"U6DHRDMH5ZOV","created_at":"2026-07-05T04:08:03.169753+00:00"},{"alias_kind":"pith_short_16","alias_value":"U6DHRDMH5ZOVSFEG","created_at":"2026-07-05T04:08:03.169753+00:00"},{"alias_kind":"pith_short_8","alias_value":"U6DHRDMH","created_at":"2026-07-05T04:08:03.169753+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":73,"internal_anchor_count":73,"sample":[{"citing_arxiv_id":"2607.07693","citing_title":"Selective Timestep Weighting and Advantage-Based Replay for Sample-Efficient Diffusion RLHF","ref_index":11,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25445","citing_title":"C3-Bench: A Context-Aware Change Captioning Benchmark","ref_index":37,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22347","citing_title":"Customizing Video Portraits via Identity-ActionDecoupling","ref_index":36,"is_internal_anchor":true},{"citing_arxiv_id":"2607.02291","citing_title":"Optimizing Visual Generative Models via Distribution-wise Rewards","ref_index":11,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12263","citing_title":"VOID: Defeating Unauthorized Mimicry in Latent Diffusion Models","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2606.07117","citing_title":"Native3D: End-to-End 3D Scene Generation via Unified Mesh-Texture Modeling and Semantic Alignment","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2607.00310","citing_title":"RetailSMV: Exocentric vs. Egocentric Adaptation of Foundation Video World Models in Retail","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03216","citing_title":"Follow-Your-Preference++: Rethinking Preference Alignment for Image Inpainting","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2606.01858","citing_title":"Polaris: Scaling Up Instruction-Guided Image Generation Towards Millions of Personalized Style Needs","ref_index":75,"is_internal_anchor":true},{"citing_arxiv_id":"2605.22050","citing_title":"Broken Memories: Detecting and Mitigating Memorization in Diffusion Models with Degraded Generations","ref_index":15,"is_internal_anchor":true},{"citing_arxiv_id":"2605.19729","citing_title":"LIFT and PLACE: A Simple, Stable, and Effective Knowledge Distillation Framework for Lightweight Diffusion Models","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2605.15055","citing_title":"DiffusionOPD: A Unified Perspective of On-Policy Distillation in Diffusion Models","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2606.31683","citing_title":"Histogram-constrained Image Generation","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2606.31026","citing_title":"OTCache: Optimal Transport for Geometry-Aware Caching in Diffusion Models","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2605.05204","citing_title":"D-OPSD: On-Policy Self-Distillation for Continuously Tuning Step-Distilled Diffusion Models","ref_index":29,"is_internal_anchor":true},{"citing_arxiv_id":"2605.20278","citing_title":"ClaimDiff-RL: Fine-Grained Caption Reinforcement Learning through Visual Claim Comparison","ref_index":11,"is_internal_anchor":true},{"citing_arxiv_id":"2605.25798","citing_title":"DiSC: Resolution-Scalable Acceleration of Diffusion Models by Exploiting Sparsity and Cached Token Reuse with Hash-based Distribution","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2605.26391","citing_title":"Garment Particles: A 2D--3D Symmetric Garment Representation for Generation and Editing","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2605.30825","citing_title":"Unlearning in Diffusion Models: A Unified Framework with KL Divergence and Likelihood Constraints","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.01481","citing_title":"SafeGen-Bench: Benchmarking Safety in Image-Conditioned Text-to-Video Generation","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22481","citing_title":"Lighting-Consistent Object Transfer Across Radiance Fields","ref_index":136,"is_internal_anchor":true},{"citing_arxiv_id":"2605.22050","citing_title":"Broken Memories: Detecting and Mitigating Memorization in Diffusion Models with Degraded Generations","ref_index":15,"is_internal_anchor":true},{"citing_arxiv_id":"2410.16431","citing_title":"Conjuring Semantic Similarity","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2412.04300","citing_title":"T2I-FactualBench: Benchmarking the Factuality of Text-to-Image Models with Knowledge-Intensive Concepts","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2502.02452","citing_title":"Personalization Toolkit: Training Free Personalization of Large Vision Language Models","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":2,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U6DHRDMH5ZOVSFEGBMUK5UUQWL","json":"https://pith.science/pith/U6DHRDMH5ZOVSFEGBMUK5UUQWL.json","graph_json":"https://pith.science/api/pith-number/U6DHRDMH5ZOVSFEGBMUK5UUQWL/graph.json","events_json":"https://pith.science/api/pith-number/U6DHRDMH5ZOVSFEGBMUK5UUQWL/events.json","paper":"https://pith.science/paper/U6DHRDMH"},"agent_actions":{"view_html":"https://pith.science/pith/U6DHRDMH5ZOVSFEGBMUK5UUQWL","download_json":"https://pith.science/pith/U6DHRDMH5ZOVSFEGBMUK5UUQWL.json","view_paper":"https://pith.science/paper/U6DHRDMH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.08718&json=true","fetch_graph":"https://pith.science/api/pith-number/U6DHRDMH5ZOVSFEGBMUK5UUQWL/graph.json","fetch_events":"https://pith.science/api/pith-number/U6DHRDMH5ZOVSFEGBMUK5UUQWL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U6DHRDMH5ZOVSFEGBMUK5UUQWL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U6DHRDMH5ZOVSFEGBMUK5UUQWL/action/storage_attestation","attest_author":"https://pith.science/pith/U6DHRDMH5ZOVSFEGBMUK5UUQWL/action/author_attestation","sign_citation":"https://pith.science/pith/U6DHRDMH5ZOVSFEGBMUK5UUQWL/action/citation_signature","submit_replication":"https://pith.science/pith/U6DHRDMH5ZOVSFEGBMUK5UUQWL/action/replication_record"}},"created_at":"2026-07-05T04:08:03.169753+00:00","updated_at":"2026-07-05T04:08:03.169753+00:00"}