{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2BLVNSH3KEMNWS2L53AA5LJSSC","short_pith_number":"pith:2BLVNSH3","schema_version":"1.0","canonical_sha256":"d05756c8fb5118db4b4beec00ead3290a7f4eecaf49cc73c30573b913510bbb7","source":{"kind":"arxiv","id":"2506.04598","version":1},"attestation_state":"computed","paper":{"title":"Scaling Laws for Robust Comparison of Open Foundation Language-Vision Models and Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Giovanni Pucceti, Jenia Jitsev, Marianna Nezhurina, Mehdi Cherti, Romain Beaumont, Tomer Porian, Tommie Kerssies","submitted_at":"2025-06-05T03:35:59Z","abstract_excerpt":"In studies of transferable learning, scaling laws are obtained for various important foundation models to predict their properties and performance at larger scales. We show here how scaling law derivation can also be used for model and dataset comparison, allowing to decide which procedure is to be preferred for pre-training. For the first time, full scaling laws based on dense measurements across a wide span of model and samples seen scales are derived for two important language-vision learning procedures, CLIP and MaMMUT, that use either contrastive only or contrastive and captioning text ge"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.04598","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-05T03:35:59Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"fba954b0739546950a7f7d695072d8fa88cc1b442af7283fb377a17f6c87ed71","abstract_canon_sha256":"18d8b88aa6e42fddeb2bf98a931caf632bacc8610a3bbe7726bfe9946ce12106"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:17.560044Z","signature_b64":"luWJVbfBYY40HZuR688xQ9CSF4wfPAZhocaZ0AjDqt+nlS00507zgQuwosOfaU0aWZzPsF6QeboGU4JInm4CCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d05756c8fb5118db4b4beec00ead3290a7f4eecaf49cc73c30573b913510bbb7","last_reissued_at":"2026-07-05T11:16:17.559445Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:17.559445Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Laws for Robust Comparison of Open Foundation Language-Vision Models and Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Giovanni Pucceti, Jenia Jitsev, Marianna Nezhurina, Mehdi Cherti, Romain Beaumont, Tomer Porian, Tommie Kerssies","submitted_at":"2025-06-05T03:35:59Z","abstract_excerpt":"In studies of transferable learning, scaling laws are obtained for various important foundation models to predict their properties and performance at larger scales. We show here how scaling law derivation can also be used for model and dataset comparison, allowing to decide which procedure is to be preferred for pre-training. For the first time, full scaling laws based on dense measurements across a wide span of model and samples seen scales are derived for two important language-vision learning procedures, CLIP and MaMMUT, that use either contrastive only or contrastive and captioning text ge"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.04598","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.04598/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.04598","created_at":"2026-07-05T11:16:17.559518+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.04598v1","created_at":"2026-07-05T11:16:17.559518+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.04598","created_at":"2026-07-05T11:16:17.559518+00:00"},{"alias_kind":"pith_short_12","alias_value":"2BLVNSH3KEMN","created_at":"2026-07-05T11:16:17.559518+00:00"},{"alias_kind":"pith_short_16","alias_value":"2BLVNSH3KEMNWS2L","created_at":"2026-07-05T11:16:17.559518+00:00"},{"alias_kind":"pith_short_8","alias_value":"2BLVNSH3","created_at":"2026-07-05T11:16:17.559518+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":222,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":222,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14629","citing_title":"Switch-KD: Visual-Switch Knowledge Distillation for Vision-Language Models","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2BLVNSH3KEMNWS2L53AA5LJSSC","json":"https://pith.science/pith/2BLVNSH3KEMNWS2L53AA5LJSSC.json","graph_json":"https://pith.science/api/pith-number/2BLVNSH3KEMNWS2L53AA5LJSSC/graph.json","events_json":"https://pith.science/api/pith-number/2BLVNSH3KEMNWS2L53AA5LJSSC/events.json","paper":"https://pith.science/paper/2BLVNSH3"},"agent_actions":{"view_html":"https://pith.science/pith/2BLVNSH3KEMNWS2L53AA5LJSSC","download_json":"https://pith.science/pith/2BLVNSH3KEMNWS2L53AA5LJSSC.json","view_paper":"https://pith.science/paper/2BLVNSH3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.04598&json=true","fetch_graph":"https://pith.science/api/pith-number/2BLVNSH3KEMNWS2L53AA5LJSSC/graph.json","fetch_events":"https://pith.science/api/pith-number/2BLVNSH3KEMNWS2L53AA5LJSSC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2BLVNSH3KEMNWS2L53AA5LJSSC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2BLVNSH3KEMNWS2L53AA5LJSSC/action/storage_attestation","attest_author":"https://pith.science/pith/2BLVNSH3KEMNWS2L53AA5LJSSC/action/author_attestation","sign_citation":"https://pith.science/pith/2BLVNSH3KEMNWS2L53AA5LJSSC/action/citation_signature","submit_replication":"https://pith.science/pith/2BLVNSH3KEMNWS2L53AA5LJSSC/action/replication_record"}},"created_at":"2026-07-05T11:16:17.559518+00:00","updated_at":"2026-07-05T11:16:17.559518+00:00"}