{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:I7LVY5JKJGWFQLHLA7LRVOTD6K","short_pith_number":"pith:I7LVY5JK","schema_version":"1.0","canonical_sha256":"47d75c752a49ac582ceb07d71aba63f2a9d4ec4f6ecbe0ae8dfa5f0c6ff8d55c","source":{"kind":"arxiv","id":"2108.02818","version":1},"attestation_state":"computed","paper":{"title":"Evaluating CLIP: Towards Characterization of Broader Capabilities and Downstream Implications","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CV","authors_text":"Alec Radford, Gretchen Krueger, Jack Clark, Jong Wook Kim, Miles Brundage, Sandhini Agarwal","submitted_at":"2021-08-05T19:05:57Z","abstract_excerpt":"Recently, there have been breakthroughs in computer vision (\"CV\") models that are more generalizable with the advent of models such as CLIP and ALIGN. In this paper, we analyze CLIP and highlight some of the challenges such models pose. CLIP reduces the need for task specific training data, potentially opening up many niche tasks to automation. CLIP also allows its users to flexibly specify image classification classes in natural language, which we find can shift how biases manifest. Additionally, through some preliminary probes we find that CLIP can inherit biases found in prior computer visi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.02818","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-08-05T19:05:57Z","cross_cats_sorted":["cs.AI","cs.CY"],"title_canon_sha256":"58c62d04a2fcef8e43001fb45290e7d2d75af5a0fca8791342aea7fa42b4d45b","abstract_canon_sha256":"b9161a8104ae8b30e41dc24d36b21f0818715f68d941f46c608d6ec6f4da4d69"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:03:42.350380Z","signature_b64":"4M61VtctdO//pSaqamwvGGYEeTqciWPstOulD1vfIXYmEWl80orhtwDtKyWGUrLQP8+baPxB+ygUIKhzGLRgCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"47d75c752a49ac582ceb07d71aba63f2a9d4ec4f6ecbe0ae8dfa5f0c6ff8d55c","last_reissued_at":"2026-07-05T03:03:42.349930Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:03:42.349930Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating CLIP: Towards Characterization of Broader Capabilities and Downstream Implications","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CV","authors_text":"Alec Radford, Gretchen Krueger, Jack Clark, Jong Wook Kim, Miles Brundage, Sandhini Agarwal","submitted_at":"2021-08-05T19:05:57Z","abstract_excerpt":"Recently, there have been breakthroughs in computer vision (\"CV\") models that are more generalizable with the advent of models such as CLIP and ALIGN. In this paper, we analyze CLIP and highlight some of the challenges such models pose. CLIP reduces the need for task specific training data, potentially opening up many niche tasks to automation. CLIP also allows its users to flexibly specify image classification classes in natural language, which we find can shift how biases manifest. Additionally, through some preliminary probes we find that CLIP can inherit biases found in prior computer visi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.02818","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.02818/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.02818","created_at":"2026-07-05T03:03:42.349983+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.02818v1","created_at":"2026-07-05T03:03:42.349983+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.02818","created_at":"2026-07-05T03:03:42.349983+00:00"},{"alias_kind":"pith_short_12","alias_value":"I7LVY5JKJGWF","created_at":"2026-07-05T03:03:42.349983+00:00"},{"alias_kind":"pith_short_16","alias_value":"I7LVY5JKJGWFQLHL","created_at":"2026-07-05T03:03:42.349983+00:00"},{"alias_kind":"pith_short_8","alias_value":"I7LVY5JK","created_at":"2026-07-05T03:03:42.349983+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19151","citing_title":"The Market in the Model: Latent Diffusion as Neural Economy","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28719","citing_title":"ComMem: Complementary Memory Systems for Test-Time Adaptation of Vision-Language Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00640","citing_title":"An Attribute-Based Measure of Video Complexity","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2104.08718","citing_title":"CLIPScore: A Reference-free Evaluation Metric for Image Captioning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13305","citing_title":"Bias at the End of the Score","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I7LVY5JKJGWFQLHLA7LRVOTD6K","json":"https://pith.science/pith/I7LVY5JKJGWFQLHLA7LRVOTD6K.json","graph_json":"https://pith.science/api/pith-number/I7LVY5JKJGWFQLHLA7LRVOTD6K/graph.json","events_json":"https://pith.science/api/pith-number/I7LVY5JKJGWFQLHLA7LRVOTD6K/events.json","paper":"https://pith.science/paper/I7LVY5JK"},"agent_actions":{"view_html":"https://pith.science/pith/I7LVY5JKJGWFQLHLA7LRVOTD6K","download_json":"https://pith.science/pith/I7LVY5JKJGWFQLHLA7LRVOTD6K.json","view_paper":"https://pith.science/paper/I7LVY5JK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.02818&json=true","fetch_graph":"https://pith.science/api/pith-number/I7LVY5JKJGWFQLHLA7LRVOTD6K/graph.json","fetch_events":"https://pith.science/api/pith-number/I7LVY5JKJGWFQLHLA7LRVOTD6K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I7LVY5JKJGWFQLHLA7LRVOTD6K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I7LVY5JKJGWFQLHLA7LRVOTD6K/action/storage_attestation","attest_author":"https://pith.science/pith/I7LVY5JKJGWFQLHLA7LRVOTD6K/action/author_attestation","sign_citation":"https://pith.science/pith/I7LVY5JKJGWFQLHLA7LRVOTD6K/action/citation_signature","submit_replication":"https://pith.science/pith/I7LVY5JKJGWFQLHLA7LRVOTD6K/action/replication_record"}},"created_at":"2026-07-05T03:03:42.349983+00:00","updated_at":"2026-07-05T03:03:42.349983+00:00"}