{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:NV7PGOB7N7SUAL4W2YN6ITG37S","short_pith_number":"pith:NV7PGOB7","schema_version":"1.0","canonical_sha256":"6d7ef3383f6fe5402f96d61be44cdbfc8e37cae5bcb9ca80d776cdb5a6821186","source":{"kind":"arxiv","id":"2204.11573","version":4},"attestation_state":"computed","paper":{"title":"Joint-Modal Label Denoising for Weakly-Supervised Audio-Visual Video Parsing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chen Qian, Hang Zhou, Haoyue Cheng, Limin Wang, Wayne Wu, Zhaoyang Liu","submitted_at":"2022-04-25T11:41:17Z","abstract_excerpt":"This paper focuses on the weakly-supervised audio-visual video parsing task, which aims to recognize all events belonging to each modality and localize their temporal boundaries. This task is challenging because only overall labels indicating the video events are provided for training. However, an event might be labeled but not appear in one of the modalities, which results in a modality-specific noisy label problem. In this work, we propose a training strategy to identify and remove modality-specific noisy labels dynamically. It is motivated by two key observations: 1) networks tend to learn "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.11573","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-04-25T11:41:17Z","cross_cats_sorted":[],"title_canon_sha256":"3b8f6a3b01af0fd7af35c6de14b451fdb123da421f9bcf566ae8387f760957f3","abstract_canon_sha256":"fb5092c2932ba4e45221ef6b6468923a0a6bb86d19b0ac682421735adf7f6a00"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:45:01.101552Z","signature_b64":"jeztuny9Am5ppWNhkDd4cCImjo5PjvFiu8YQwKNqCd2elWnc9oMF/QcRoZcTL2F3b0N1yNc0AWSISN26KxZsBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6d7ef3383f6fe5402f96d61be44cdbfc8e37cae5bcb9ca80d776cdb5a6821186","last_reissued_at":"2026-07-05T04:45:01.101124Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:45:01.101124Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Joint-Modal Label Denoising for Weakly-Supervised Audio-Visual Video Parsing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chen Qian, Hang Zhou, Haoyue Cheng, Limin Wang, Wayne Wu, Zhaoyang Liu","submitted_at":"2022-04-25T11:41:17Z","abstract_excerpt":"This paper focuses on the weakly-supervised audio-visual video parsing task, which aims to recognize all events belonging to each modality and localize their temporal boundaries. This task is challenging because only overall labels indicating the video events are provided for training. However, an event might be labeled but not appear in one of the modalities, which results in a modality-specific noisy label problem. In this work, we propose a training strategy to identify and remove modality-specific noisy labels dynamically. It is motivated by two key observations: 1) networks tend to learn "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.11573","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.11573/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.11573","created_at":"2026-07-05T04:45:01.101180+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.11573v4","created_at":"2026-07-05T04:45:01.101180+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.11573","created_at":"2026-07-05T04:45:01.101180+00:00"},{"alias_kind":"pith_short_12","alias_value":"NV7PGOB7N7SU","created_at":"2026-07-05T04:45:01.101180+00:00"},{"alias_kind":"pith_short_16","alias_value":"NV7PGOB7N7SUAL4W","created_at":"2026-07-05T04:45:01.101180+00:00"},{"alias_kind":"pith_short_8","alias_value":"NV7PGOB7","created_at":"2026-07-05T04:45:01.101180+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NV7PGOB7N7SUAL4W2YN6ITG37S","json":"https://pith.science/pith/NV7PGOB7N7SUAL4W2YN6ITG37S.json","graph_json":"https://pith.science/api/pith-number/NV7PGOB7N7SUAL4W2YN6ITG37S/graph.json","events_json":"https://pith.science/api/pith-number/NV7PGOB7N7SUAL4W2YN6ITG37S/events.json","paper":"https://pith.science/paper/NV7PGOB7"},"agent_actions":{"view_html":"https://pith.science/pith/NV7PGOB7N7SUAL4W2YN6ITG37S","download_json":"https://pith.science/pith/NV7PGOB7N7SUAL4W2YN6ITG37S.json","view_paper":"https://pith.science/paper/NV7PGOB7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.11573&json=true","fetch_graph":"https://pith.science/api/pith-number/NV7PGOB7N7SUAL4W2YN6ITG37S/graph.json","fetch_events":"https://pith.science/api/pith-number/NV7PGOB7N7SUAL4W2YN6ITG37S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NV7PGOB7N7SUAL4W2YN6ITG37S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NV7PGOB7N7SUAL4W2YN6ITG37S/action/storage_attestation","attest_author":"https://pith.science/pith/NV7PGOB7N7SUAL4W2YN6ITG37S/action/author_attestation","sign_citation":"https://pith.science/pith/NV7PGOB7N7SUAL4W2YN6ITG37S/action/citation_signature","submit_replication":"https://pith.science/pith/NV7PGOB7N7SUAL4W2YN6ITG37S/action/replication_record"}},"created_at":"2026-07-05T04:45:01.101180+00:00","updated_at":"2026-07-05T04:45:01.101180+00:00"}