{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:K4EXCOGNRWF7L7ZD43KXSA5W6X","short_pith_number":"pith:K4EXCOGN","schema_version":"1.0","canonical_sha256":"57097138cd8d8bf5ff23e6d57903b6f5dd9ee07f01dd9e6af8f31440686f43ae","source":{"kind":"arxiv","id":"2312.06726","version":4},"attestation_state":"computed","paper":{"title":"Filter & Align: Leveraging Human Knowledge to Curate Image-Text Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cihang Xie, Fangxun Shu, Hao Jiang, Lei Zhang, Sucheng Ren, Tianyang Liu","submitted_at":"2023-12-11T05:57:09Z","abstract_excerpt":"The increasing availability of image-text pairs has largely fueled the rapid advancement in vision-language foundation models. However, the vast scale of these datasets inevitably introduces significant variability in data quality, which can adversely affect the model performance. This highlights the critical role of data filtering, not only to enhance training efficiency but also to improve overall data quality. Existing methods typically rely on metrics such as CLIP Score and BLIP Score, which are derived from pre-trained models. However, these models are often trained on uncurated, noisy da"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.06726","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-11T05:57:09Z","cross_cats_sorted":[],"title_canon_sha256":"980618c5d7d3a0b959fd196832d46f0049043abc938a540a80d82dc3176ec21a","abstract_canon_sha256":"ba58f30364a36d9e570e1377c7c4ddfc83ac53d46fac6d28e8650d59b86781e8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:02:55.577765Z","signature_b64":"/asTwbEieEwJ7AqzRcXaYty5EZ0j8v7hfSiC1cROxr93Cy7QOq1EvOHa717Q1Roo6eSgxhPxHwijqMAjHyr4Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"57097138cd8d8bf5ff23e6d57903b6f5dd9ee07f01dd9e6af8f31440686f43ae","last_reissued_at":"2026-07-05T09:02:55.577302Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:02:55.577302Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Filter & Align: Leveraging Human Knowledge to Curate Image-Text Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cihang Xie, Fangxun Shu, Hao Jiang, Lei Zhang, Sucheng Ren, Tianyang Liu","submitted_at":"2023-12-11T05:57:09Z","abstract_excerpt":"The increasing availability of image-text pairs has largely fueled the rapid advancement in vision-language foundation models. However, the vast scale of these datasets inevitably introduces significant variability in data quality, which can adversely affect the model performance. This highlights the critical role of data filtering, not only to enhance training efficiency but also to improve overall data quality. Existing methods typically rely on metrics such as CLIP Score and BLIP Score, which are derived from pre-trained models. However, these models are often trained on uncurated, noisy da"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.06726","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.06726/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.06726","created_at":"2026-07-05T09:02:55.577371+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.06726v4","created_at":"2026-07-05T09:02:55.577371+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.06726","created_at":"2026-07-05T09:02:55.577371+00:00"},{"alias_kind":"pith_short_12","alias_value":"K4EXCOGNRWF7","created_at":"2026-07-05T09:02:55.577371+00:00"},{"alias_kind":"pith_short_16","alias_value":"K4EXCOGNRWF7L7ZD","created_at":"2026-07-05T09:02:55.577371+00:00"},{"alias_kind":"pith_short_8","alias_value":"K4EXCOGN","created_at":"2026-07-05T09:02:55.577371+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K4EXCOGNRWF7L7ZD43KXSA5W6X","json":"https://pith.science/pith/K4EXCOGNRWF7L7ZD43KXSA5W6X.json","graph_json":"https://pith.science/api/pith-number/K4EXCOGNRWF7L7ZD43KXSA5W6X/graph.json","events_json":"https://pith.science/api/pith-number/K4EXCOGNRWF7L7ZD43KXSA5W6X/events.json","paper":"https://pith.science/paper/K4EXCOGN"},"agent_actions":{"view_html":"https://pith.science/pith/K4EXCOGNRWF7L7ZD43KXSA5W6X","download_json":"https://pith.science/pith/K4EXCOGNRWF7L7ZD43KXSA5W6X.json","view_paper":"https://pith.science/paper/K4EXCOGN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.06726&json=true","fetch_graph":"https://pith.science/api/pith-number/K4EXCOGNRWF7L7ZD43KXSA5W6X/graph.json","fetch_events":"https://pith.science/api/pith-number/K4EXCOGNRWF7L7ZD43KXSA5W6X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K4EXCOGNRWF7L7ZD43KXSA5W6X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K4EXCOGNRWF7L7ZD43KXSA5W6X/action/storage_attestation","attest_author":"https://pith.science/pith/K4EXCOGNRWF7L7ZD43KXSA5W6X/action/author_attestation","sign_citation":"https://pith.science/pith/K4EXCOGNRWF7L7ZD43KXSA5W6X/action/citation_signature","submit_replication":"https://pith.science/pith/K4EXCOGNRWF7L7ZD43KXSA5W6X/action/replication_record"}},"created_at":"2026-07-05T09:02:55.577371+00:00","updated_at":"2026-07-05T09:02:55.577371+00:00"}