{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FAMIDUGL2O4EZ7APXKRM7UDJUA","short_pith_number":"pith:FAMIDUGL","schema_version":"1.0","canonical_sha256":"281881d0cbd3b84cfc0fbaa2cfd069a01e62d4a2ed3b95af8b230a19d9c77f45","source":{"kind":"arxiv","id":"2402.02055","version":1},"attestation_state":"computed","paper":{"title":"Variance Alignment Score: A Simple But Tough-to-Beat Data Selection Method for Multimodal Contrastive Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Kevin Jamieson, Simon Shaolei Du, Wendan Yan, Yifang Chen, Yiping Wang","submitted_at":"2024-02-03T06:29:04Z","abstract_excerpt":"In recent years, data selection has emerged as a core issue for large-scale visual-language model pretraining, especially on noisy web-curated datasets. One widely adopted strategy assigns quality scores such as CLIP similarity for each sample and retains the data pairs with the highest scores. However, these approaches are agnostic of data distribution and always fail to select the most informative samples. To solve this problem, we propose a simple yet theoretically principled metric named Variance Alignment Score (VAS), which has the form $\\langle \\Sigma_{\\text{test}}, \\Sigma_i\\rangle$. Her"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.02055","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-03T06:29:04Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"77fca2283aae9ccb48314e47990499ad42997f16907c28d0f9c73d71dcfcac09","abstract_canon_sha256":"f60f3e93c8c0113c7779e18873a7001876d5d6c4d6beb9ad6d6942382cc11943"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:41:10.262750Z","signature_b64":"0scS3bvcAnFIWQ7EuGJ4qKqwvEPQIgwZNPasZQURTlNxkL6JG2pQCq5cDEbusAFSTgSJvAocj4My5EgBqHeGAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"281881d0cbd3b84cfc0fbaa2cfd069a01e62d4a2ed3b95af8b230a19d9c77f45","last_reissued_at":"2026-07-05T07:41:10.262285Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:41:10.262285Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Variance Alignment Score: A Simple But Tough-to-Beat Data Selection Method for Multimodal Contrastive Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Kevin Jamieson, Simon Shaolei Du, Wendan Yan, Yifang Chen, Yiping Wang","submitted_at":"2024-02-03T06:29:04Z","abstract_excerpt":"In recent years, data selection has emerged as a core issue for large-scale visual-language model pretraining, especially on noisy web-curated datasets. One widely adopted strategy assigns quality scores such as CLIP similarity for each sample and retains the data pairs with the highest scores. However, these approaches are agnostic of data distribution and always fail to select the most informative samples. To solve this problem, we propose a simple yet theoretically principled metric named Variance Alignment Score (VAS), which has the form $\\langle \\Sigma_{\\text{test}}, \\Sigma_i\\rangle$. Her"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.02055","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.02055/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.02055","created_at":"2026-07-05T07:41:10.262347+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.02055v1","created_at":"2026-07-05T07:41:10.262347+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.02055","created_at":"2026-07-05T07:41:10.262347+00:00"},{"alias_kind":"pith_short_12","alias_value":"FAMIDUGL2O4E","created_at":"2026-07-05T07:41:10.262347+00:00"},{"alias_kind":"pith_short_16","alias_value":"FAMIDUGL2O4EZ7AP","created_at":"2026-07-05T07:41:10.262347+00:00"},{"alias_kind":"pith_short_8","alias_value":"FAMIDUGL","created_at":"2026-07-05T07:41:10.262347+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.10550","citing_title":"ContextRefine-CLIP for EPIC-KITCHENS-100 Multi-Instance Retrieval Challenge 2025","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FAMIDUGL2O4EZ7APXKRM7UDJUA","json":"https://pith.science/pith/FAMIDUGL2O4EZ7APXKRM7UDJUA.json","graph_json":"https://pith.science/api/pith-number/FAMIDUGL2O4EZ7APXKRM7UDJUA/graph.json","events_json":"https://pith.science/api/pith-number/FAMIDUGL2O4EZ7APXKRM7UDJUA/events.json","paper":"https://pith.science/paper/FAMIDUGL"},"agent_actions":{"view_html":"https://pith.science/pith/FAMIDUGL2O4EZ7APXKRM7UDJUA","download_json":"https://pith.science/pith/FAMIDUGL2O4EZ7APXKRM7UDJUA.json","view_paper":"https://pith.science/paper/FAMIDUGL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.02055&json=true","fetch_graph":"https://pith.science/api/pith-number/FAMIDUGL2O4EZ7APXKRM7UDJUA/graph.json","fetch_events":"https://pith.science/api/pith-number/FAMIDUGL2O4EZ7APXKRM7UDJUA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FAMIDUGL2O4EZ7APXKRM7UDJUA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FAMIDUGL2O4EZ7APXKRM7UDJUA/action/storage_attestation","attest_author":"https://pith.science/pith/FAMIDUGL2O4EZ7APXKRM7UDJUA/action/author_attestation","sign_citation":"https://pith.science/pith/FAMIDUGL2O4EZ7APXKRM7UDJUA/action/citation_signature","submit_replication":"https://pith.science/pith/FAMIDUGL2O4EZ7APXKRM7UDJUA/action/replication_record"}},"created_at":"2026-07-05T07:41:10.262347+00:00","updated_at":"2026-07-05T07:41:10.262347+00:00"}