{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:PMF4WBSRQLHHFHYSCLGPY4HND3","short_pith_number":"pith:PMF4WBSR","schema_version":"1.0","canonical_sha256":"7b0bcb065182ce729f1212ccfc70ed1ec9257b161817588c5a22711f2257eead","source":{"kind":"arxiv","id":"2102.05918","version":2},"attestation_state":"computed","paper":{"title":"Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chao Jia, Hieu Pham, Quoc V. Le, Tom Duerig, Ye Xia, Yinfei Yang, Yi-Ting Chen, Yunhsuan Sung, Zarana Parekh, Zhen Li","submitted_at":"2021-02-11T10:08:12Z","abstract_excerpt":"Pre-trained representations are becoming crucial for many NLP and perception tasks. While representation learning in NLP has transitioned to training on raw text without human annotations, visual and vision-language representations still rely heavily on curated training datasets that are expensive or require expert knowledge. For vision applications, representations are mostly learned using datasets with explicit class labels such as ImageNet or OpenImages. For vision-language, popular datasets like Conceptual Captions, MSCOCO, or CLIP all involve a non-trivial data collection (and cleaning) p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.05918","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-02-11T10:08:12Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"014b1ee0d362b8ad566293f27262f7584eb3fae1619db2fbf94fe78709138490","abstract_canon_sha256":"d00fe77f2ddaf8c54ef23758c3a5ff3a82d546ddd1544a89cedb9b606d7a7939"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:48:21.053136Z","signature_b64":"NkhnZDN/XlA45VXW9PRk/xQ/+icnX7Hy+3qVZxYOg0Clkk1URVzOul15U0bSpn/wdE2Js6pBRLTR6l8YtyfIBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7b0bcb065182ce729f1212ccfc70ed1ec9257b161817588c5a22711f2257eead","last_reissued_at":"2026-07-05T02:48:21.052575Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:48:21.052575Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chao Jia, Hieu Pham, Quoc V. Le, Tom Duerig, Ye Xia, Yinfei Yang, Yi-Ting Chen, Yunhsuan Sung, Zarana Parekh, Zhen Li","submitted_at":"2021-02-11T10:08:12Z","abstract_excerpt":"Pre-trained representations are becoming crucial for many NLP and perception tasks. While representation learning in NLP has transitioned to training on raw text without human annotations, visual and vision-language representations still rely heavily on curated training datasets that are expensive or require expert knowledge. For vision applications, representations are mostly learned using datasets with explicit class labels such as ImageNet or OpenImages. For vision-language, popular datasets like Conceptual Captions, MSCOCO, or CLIP all involve a non-trivial data collection (and cleaning) p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.05918","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.05918/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.05918","created_at":"2026-07-05T02:48:21.052640+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.05918v2","created_at":"2026-07-05T02:48:21.052640+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.05918","created_at":"2026-07-05T02:48:21.052640+00:00"},{"alias_kind":"pith_short_12","alias_value":"PMF4WBSRQLHH","created_at":"2026-07-05T02:48:21.052640+00:00"},{"alias_kind":"pith_short_16","alias_value":"PMF4WBSRQLHHFHYS","created_at":"2026-07-05T02:48:21.052640+00:00"},{"alias_kind":"pith_short_8","alias_value":"PMF4WBSR","created_at":"2026-07-05T02:48:21.052640+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17030","citing_title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02484","citing_title":"Combating Textual Noise and Redundancy: Entropy-Aware Dense Visual Token Pruning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11074","citing_title":"Modeling Complex Behaviors: Multi-Personality Composition and Dynamic Switching in Vision-Language Models","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09881","citing_title":"Toward Calibrated, Fair, and accurate Deepfake Detection","ref_index":266,"is_internal_anchor":false},{"citing_arxiv_id":"2208.14649","citing_title":"DetailCLIP: Injecting Image Details into CLIP's Feature Space","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2603.11689","citing_title":"Explicit Logic Channel for Validation and Enhancement of MLLMs on Zero-Shot Tasks","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19410","citing_title":"Vision Harnessing Agent for Open Ad-hoc Segmentation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2508.09691","citing_title":"PaCo-FR: Patch-Pixel Aligned End-to-End Codebook Learning for Facial Representation Pre-training","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2305.07895","citing_title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2111.11432","citing_title":"Florence: A New Foundation Model for Computer Vision","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09921","citing_title":"WikiCLIP: An Efficient Contrastive Baseline for Open-domain Visual Entity Recognition","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02753","citing_title":"DeCo-DETR: Decoupled Cognition DETR for efficient Open-Vocabulary Object Detection","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02753","citing_title":"DeCo-DETR: Decoupled Cognition DETR for efficient Open-Vocabulary Object Detection","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2210.08402","citing_title":"LAION-5B: An open large-scale dataset for training next generation image-text models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2212.04089","citing_title":"Editing Models with Task Arithmetic","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2111.02114","citing_title":"LAION-400M: Open Dataset of CLIP-Filtered 400 Million Image-Text Pairs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2204.14198","citing_title":"Flamingo: a Visual Language Model for Few-Shot Learning","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2301.12597","citing_title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07802","citing_title":"Latent Anomaly Knowledge Excavation: Unveiling Sparse Sensitive Neurons in Vision-Language Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01048","citing_title":"Compared to What? Baselines and Metrics for Counterfactual Prompting","ref_index":93,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PMF4WBSRQLHHFHYSCLGPY4HND3","json":"https://pith.science/pith/PMF4WBSRQLHHFHYSCLGPY4HND3.json","graph_json":"https://pith.science/api/pith-number/PMF4WBSRQLHHFHYSCLGPY4HND3/graph.json","events_json":"https://pith.science/api/pith-number/PMF4WBSRQLHHFHYSCLGPY4HND3/events.json","paper":"https://pith.science/paper/PMF4WBSR"},"agent_actions":{"view_html":"https://pith.science/pith/PMF4WBSRQLHHFHYSCLGPY4HND3","download_json":"https://pith.science/pith/PMF4WBSRQLHHFHYSCLGPY4HND3.json","view_paper":"https://pith.science/paper/PMF4WBSR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.05918&json=true","fetch_graph":"https://pith.science/api/pith-number/PMF4WBSRQLHHFHYSCLGPY4HND3/graph.json","fetch_events":"https://pith.science/api/pith-number/PMF4WBSRQLHHFHYSCLGPY4HND3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PMF4WBSRQLHHFHYSCLGPY4HND3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PMF4WBSRQLHHFHYSCLGPY4HND3/action/storage_attestation","attest_author":"https://pith.science/pith/PMF4WBSRQLHHFHYSCLGPY4HND3/action/author_attestation","sign_citation":"https://pith.science/pith/PMF4WBSRQLHHFHYSCLGPY4HND3/action/citation_signature","submit_replication":"https://pith.science/pith/PMF4WBSRQLHHFHYSCLGPY4HND3/action/replication_record"}},"created_at":"2026-07-05T02:48:21.052640+00:00","updated_at":"2026-07-05T02:48:21.052640+00:00"}