{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TOKO2W372BBI46UDQYRYXR7W52","short_pith_number":"pith:TOKO2W37","schema_version":"1.0","canonical_sha256":"9b94ed5b7fd0428e7a8386238bc7f6eeb124b616d8c08a8ce39d1cfbbc49b6e6","source":{"kind":"arxiv","id":"2504.01017","version":1},"attestation_state":"computed","paper":{"title":"Scaling Language-Free Visual Representation Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Amir Bar, David Fan, Jiachen Zhu, Koustuv Sinha, Michael Rabbat, Nicolas Ballas, Saining Xie, Shengbang Tong, Xinlei Chen, Yann LeCun, Zhuang Liu","submitted_at":"2025-04-01T17:59:15Z","abstract_excerpt":"Visual Self-Supervised Learning (SSL) currently underperforms Contrastive Language-Image Pretraining (CLIP) in multimodal settings such as Visual Question Answering (VQA). This multimodal gap is often attributed to the semantics introduced by language supervision, even though visual SSL and CLIP models are often trained on different data. In this work, we ask the question: \"Do visual self-supervised approaches lag behind CLIP due to the lack of language supervision, or differences in the training data?\" We study this question by training both visual SSL and CLIP models on the same MetaCLIP dat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.01017","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-04-01T17:59:15Z","cross_cats_sorted":[],"title_canon_sha256":"531dd330054e9ed357c8d0e3d71e9f60665923494e8c763df09090aade36b1fd","abstract_canon_sha256":"627555696731a4b46b08e47e6d4ece58703e6edcf48cd62f5981a527cf33d317"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:42:54.986213Z","signature_b64":"0QY2nU9P8yb7mqDdgu3KivglXDpLBCR8hZNQfvRrHmUBGo6GS+rujnWMdjPSa2XwFzn+3+1omZZC5O90nRI4Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9b94ed5b7fd0428e7a8386238bc7f6eeb124b616d8c08a8ce39d1cfbbc49b6e6","last_reissued_at":"2026-07-05T10:42:54.985617Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:42:54.985617Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Language-Free Visual Representation Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Amir Bar, David Fan, Jiachen Zhu, Koustuv Sinha, Michael Rabbat, Nicolas Ballas, Saining Xie, Shengbang Tong, Xinlei Chen, Yann LeCun, Zhuang Liu","submitted_at":"2025-04-01T17:59:15Z","abstract_excerpt":"Visual Self-Supervised Learning (SSL) currently underperforms Contrastive Language-Image Pretraining (CLIP) in multimodal settings such as Visual Question Answering (VQA). This multimodal gap is often attributed to the semantics introduced by language supervision, even though visual SSL and CLIP models are often trained on different data. In this work, we ask the question: \"Do visual self-supervised approaches lag behind CLIP due to the lack of language supervision, or differences in the training data?\" We study this question by training both visual SSL and CLIP models on the same MetaCLIP dat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.01017","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.01017/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.01017","created_at":"2026-07-05T10:42:54.985696+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.01017v1","created_at":"2026-07-05T10:42:54.985696+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.01017","created_at":"2026-07-05T10:42:54.985696+00:00"},{"alias_kind":"pith_short_12","alias_value":"TOKO2W372BBI","created_at":"2026-07-05T10:42:54.985696+00:00"},{"alias_kind":"pith_short_16","alias_value":"TOKO2W372BBI46UD","created_at":"2026-07-05T10:42:54.985696+00:00"},{"alias_kind":"pith_short_8","alias_value":"TOKO2W37","created_at":"2026-07-05T10:42:54.985696+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.22192","citing_title":"CharTide: Data-Centric Chart-to-Code Generation via Tri-Perspective Tuning and Inquiry-Driven Evolution","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24888","citing_title":"DiffusionBench: On Holistic Evaluation of Diffusion Transformers","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12215","citing_title":"MLT-Dedup: Efficient Large-Scale Online Video Deduplication via Multi-Level Representations and Spatial-Temporal Matching","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2503.14998","citing_title":"Tables Guide Vision: Learning to See the Heart through Tabular Data","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15616","citing_title":"LENS: Multi-level Evaluation of Multimodal Reasoning with Large Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18324","citing_title":"Improved Baselines with Representation Autoencoders","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2507.14137","citing_title":"Franca: Nested Matryoshka Clustering for Scalable Visual Representation Learning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2511.08544","citing_title":"LeJEPA: Provable and Scalable Self-Supervised Learning Without the Heuristics","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2504.13181","citing_title":"Perception Encoder: The best visual embeddings are not at the output of the network","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03231","citing_title":"CoME-VL: Scaling Complementary Multi-Encoder Vision-Language Learning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11870","citing_title":"Information theoretic underpinning of self-supervised learning by clustering","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22192","citing_title":"CharTide: Data-Centric Chart-to-Code Generation via Tri-Perspective Tuning and Inquiry-Driven Evolution","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12012","citing_title":"TIPSv2: Advancing Vision-Language Pretraining with Enhanced Patch-Text Alignment","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09985","citing_title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TOKO2W372BBI46UDQYRYXR7W52","json":"https://pith.science/pith/TOKO2W372BBI46UDQYRYXR7W52.json","graph_json":"https://pith.science/api/pith-number/TOKO2W372BBI46UDQYRYXR7W52/graph.json","events_json":"https://pith.science/api/pith-number/TOKO2W372BBI46UDQYRYXR7W52/events.json","paper":"https://pith.science/paper/TOKO2W37"},"agent_actions":{"view_html":"https://pith.science/pith/TOKO2W372BBI46UDQYRYXR7W52","download_json":"https://pith.science/pith/TOKO2W372BBI46UDQYRYXR7W52.json","view_paper":"https://pith.science/paper/TOKO2W37","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.01017&json=true","fetch_graph":"https://pith.science/api/pith-number/TOKO2W372BBI46UDQYRYXR7W52/graph.json","fetch_events":"https://pith.science/api/pith-number/TOKO2W372BBI46UDQYRYXR7W52/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TOKO2W372BBI46UDQYRYXR7W52/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TOKO2W372BBI46UDQYRYXR7W52/action/storage_attestation","attest_author":"https://pith.science/pith/TOKO2W372BBI46UDQYRYXR7W52/action/author_attestation","sign_citation":"https://pith.science/pith/TOKO2W372BBI46UDQYRYXR7W52/action/citation_signature","submit_replication":"https://pith.science/pith/TOKO2W372BBI46UDQYRYXR7W52/action/replication_record"}},"created_at":"2026-07-05T10:42:54.985696+00:00","updated_at":"2026-07-05T10:42:54.985696+00:00"}