{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OEKBYRAV4ZLODQMCQCJP5EA254","short_pith_number":"pith:OEKBYRAV","schema_version":"1.0","canonical_sha256":"71141c4415e656e1c1828092fe901aef0cc5cacefc5879e0ab2b883cca0a4f07","source":{"kind":"arxiv","id":"2403.14402","version":2},"attestation_state":"computed","paper":{"title":"XLAVS-R: Cross-Lingual Audio-Visual Speech Representation Learning for Noise-Robust Speech Perception","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Bowen Shi, Changhan Wang, HyoJung Han, Juan Pino, Marine Carpuat, Mohamed Anwar, Wei-Ning Hsu","submitted_at":"2024-03-21T13:52:17Z","abstract_excerpt":"Speech recognition and translation systems perform poorly on noisy inputs, which are frequent in realistic environments. Augmenting these systems with visual signals has the potential to improve robustness to noise. However, audio-visual (AV) data is only available in limited amounts and for fewer languages than audio-only resources. To address this gap, we present XLAVS-R, a cross-lingual audio-visual speech representation model for noise-robust speech recognition and translation in over 100 languages. It is designed to maximize the benefits of limited multilingual AV pre-training data, by bu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.14402","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.SD","submitted_at":"2024-03-21T13:52:17Z","cross_cats_sorted":["cs.CL","eess.AS"],"title_canon_sha256":"0302e9c8f4f9847705bc23e49b4eecd4483b7fa1d55d81c86c988136891c953f","abstract_canon_sha256":"1994895b107f7903bff55a13f008411346514ff4388df87fa220918bbb108cec"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:54:08.891800Z","signature_b64":"9vArv2+QVmJK1jd1cai7Lc1jpN6hXfFHR8FKh/5NatGGcjjYxXxH6aibFEsl/OwSyChzMBnCdTOZoHfmN9OwDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"71141c4415e656e1c1828092fe901aef0cc5cacefc5879e0ab2b883cca0a4f07","last_reissued_at":"2026-07-05T08:54:08.891251Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:54:08.891251Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"XLAVS-R: Cross-Lingual Audio-Visual Speech Representation Learning for Noise-Robust Speech Perception","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Bowen Shi, Changhan Wang, HyoJung Han, Juan Pino, Marine Carpuat, Mohamed Anwar, Wei-Ning Hsu","submitted_at":"2024-03-21T13:52:17Z","abstract_excerpt":"Speech recognition and translation systems perform poorly on noisy inputs, which are frequent in realistic environments. Augmenting these systems with visual signals has the potential to improve robustness to noise. However, audio-visual (AV) data is only available in limited amounts and for fewer languages than audio-only resources. To address this gap, we present XLAVS-R, a cross-lingual audio-visual speech representation model for noise-robust speech recognition and translation in over 100 languages. It is designed to maximize the benefits of limited multilingual AV pre-training data, by bu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.14402","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.14402/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.14402","created_at":"2026-07-05T08:54:08.891312+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.14402v2","created_at":"2026-07-05T08:54:08.891312+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.14402","created_at":"2026-07-05T08:54:08.891312+00:00"},{"alias_kind":"pith_short_12","alias_value":"OEKBYRAV4ZLO","created_at":"2026-07-05T08:54:08.891312+00:00"},{"alias_kind":"pith_short_16","alias_value":"OEKBYRAV4ZLODQMC","created_at":"2026-07-05T08:54:08.891312+00:00"},{"alias_kind":"pith_short_8","alias_value":"OEKBYRAV","created_at":"2026-07-05T08:54:08.891312+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.07102","citing_title":"AdaCS: Adaptive Normalization for Enhanced Code-Switching ASR","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OEKBYRAV4ZLODQMCQCJP5EA254","json":"https://pith.science/pith/OEKBYRAV4ZLODQMCQCJP5EA254.json","graph_json":"https://pith.science/api/pith-number/OEKBYRAV4ZLODQMCQCJP5EA254/graph.json","events_json":"https://pith.science/api/pith-number/OEKBYRAV4ZLODQMCQCJP5EA254/events.json","paper":"https://pith.science/paper/OEKBYRAV"},"agent_actions":{"view_html":"https://pith.science/pith/OEKBYRAV4ZLODQMCQCJP5EA254","download_json":"https://pith.science/pith/OEKBYRAV4ZLODQMCQCJP5EA254.json","view_paper":"https://pith.science/paper/OEKBYRAV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.14402&json=true","fetch_graph":"https://pith.science/api/pith-number/OEKBYRAV4ZLODQMCQCJP5EA254/graph.json","fetch_events":"https://pith.science/api/pith-number/OEKBYRAV4ZLODQMCQCJP5EA254/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OEKBYRAV4ZLODQMCQCJP5EA254/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OEKBYRAV4ZLODQMCQCJP5EA254/action/storage_attestation","attest_author":"https://pith.science/pith/OEKBYRAV4ZLODQMCQCJP5EA254/action/author_attestation","sign_citation":"https://pith.science/pith/OEKBYRAV4ZLODQMCQCJP5EA254/action/citation_signature","submit_replication":"https://pith.science/pith/OEKBYRAV4ZLODQMCQCJP5EA254/action/replication_record"}},"created_at":"2026-07-05T08:54:08.891312+00:00","updated_at":"2026-07-05T08:54:08.891312+00:00"}