{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EFW4FIVXWREIMH27TCBAFH27LF","short_pith_number":"pith:EFW4FIVX","schema_version":"1.0","canonical_sha256":"216dc2a2b7b448861f5f9882029f5f596f66b4b137674c21e1309ca9dc794aed","source":{"kind":"arxiv","id":"2407.18941","version":2},"attestation_state":"computed","paper":{"title":"LEMoN: Label Error Detection using Multimodal Neighbors","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Aparna Balagopalan, Haoran Zhang, Hyewon Jeong, Jiacheng Zhu, Marzyeh Ghassemi, Nassim Oufattole, Yan Wu","submitted_at":"2024-07-10T19:36:30Z","abstract_excerpt":"Large repositories of image-caption pairs are essential for the development of vision-language models. However, these datasets are often extracted from noisy data scraped from the web, and contain many mislabeled instances. In order to improve the reliability of downstream models, it is important to identify and filter images with incorrect captions. However, beyond filtering based on image-caption embedding similarity, no prior works have proposed other methods to filter noisy multimodal data, or concretely assessed the impact of noisy captioning data on downstream training. In this work, we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.18941","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-10T19:36:30Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e0fd0b2ee00c4a198b06824c28cc7d206646f8f189223710b7d73a1b787c16bf","abstract_canon_sha256":"6eb195270adf788c6c1ec96e45f5652e2a5ff82a4808c885abc01fc938c9480e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:02.943739Z","signature_b64":"1eoIN8PWVE0qyqxXKrF5yiuI1dytbGy0AySKB45hpR09QFQpfyKH0HKNDW5H76GPqyaHHnfrQqqPtVPEwa0ZAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"216dc2a2b7b448861f5f9882029f5f596f66b4b137674c21e1309ca9dc794aed","last_reissued_at":"2026-07-05T11:16:02.943169Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:02.943169Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LEMoN: Label Error Detection using Multimodal Neighbors","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Aparna Balagopalan, Haoran Zhang, Hyewon Jeong, Jiacheng Zhu, Marzyeh Ghassemi, Nassim Oufattole, Yan Wu","submitted_at":"2024-07-10T19:36:30Z","abstract_excerpt":"Large repositories of image-caption pairs are essential for the development of vision-language models. However, these datasets are often extracted from noisy data scraped from the web, and contain many mislabeled instances. In order to improve the reliability of downstream models, it is important to identify and filter images with incorrect captions. However, beyond filtering based on image-caption embedding similarity, no prior works have proposed other methods to filter noisy multimodal data, or concretely assessed the impact of noisy captioning data on downstream training. In this work, we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.18941","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.18941/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.18941","created_at":"2026-07-05T11:16:02.943225+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.18941v2","created_at":"2026-07-05T11:16:02.943225+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.18941","created_at":"2026-07-05T11:16:02.943225+00:00"},{"alias_kind":"pith_short_12","alias_value":"EFW4FIVXWREI","created_at":"2026-07-05T11:16:02.943225+00:00"},{"alias_kind":"pith_short_16","alias_value":"EFW4FIVXWREIMH27","created_at":"2026-07-05T11:16:02.943225+00:00"},{"alias_kind":"pith_short_8","alias_value":"EFW4FIVX","created_at":"2026-07-05T11:16:02.943225+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09648","citing_title":"ArtiFact: A Large-Scale Multi-Modal Cultural Heritage Dataset","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00571","citing_title":"On the Difficulty of Learning a Meta-network for Training Data Selection","ref_index":97,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EFW4FIVXWREIMH27TCBAFH27LF","json":"https://pith.science/pith/EFW4FIVXWREIMH27TCBAFH27LF.json","graph_json":"https://pith.science/api/pith-number/EFW4FIVXWREIMH27TCBAFH27LF/graph.json","events_json":"https://pith.science/api/pith-number/EFW4FIVXWREIMH27TCBAFH27LF/events.json","paper":"https://pith.science/paper/EFW4FIVX"},"agent_actions":{"view_html":"https://pith.science/pith/EFW4FIVXWREIMH27TCBAFH27LF","download_json":"https://pith.science/pith/EFW4FIVXWREIMH27TCBAFH27LF.json","view_paper":"https://pith.science/paper/EFW4FIVX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.18941&json=true","fetch_graph":"https://pith.science/api/pith-number/EFW4FIVXWREIMH27TCBAFH27LF/graph.json","fetch_events":"https://pith.science/api/pith-number/EFW4FIVXWREIMH27TCBAFH27LF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EFW4FIVXWREIMH27TCBAFH27LF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EFW4FIVXWREIMH27TCBAFH27LF/action/storage_attestation","attest_author":"https://pith.science/pith/EFW4FIVXWREIMH27TCBAFH27LF/action/author_attestation","sign_citation":"https://pith.science/pith/EFW4FIVXWREIMH27TCBAFH27LF/action/citation_signature","submit_replication":"https://pith.science/pith/EFW4FIVXWREIMH27TCBAFH27LF/action/replication_record"}},"created_at":"2026-07-05T11:16:02.943225+00:00","updated_at":"2026-07-05T11:16:02.943225+00:00"}