{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BGWWX67DWVYM4M464L2QFL7EAT","short_pith_number":"pith:BGWWX67D","schema_version":"1.0","canonical_sha256":"09ad6bfbe3b570ce339ee2f502afe404fe28c921eff47aaeb8fa46e7ff212e0c","source":{"kind":"arxiv","id":"2409.12370","version":1},"attestation_state":"computed","paper":{"title":"Robust Audiovisual Speech Recognition Models with Mixture-of-Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.SD"],"primary_cat":"eess.AS","authors_text":"Ruihua Song, Shinji Watanabe, Xuankai Chang, Yichen Lu, Yifan Peng, Yihan Wu","submitted_at":"2024-09-19T00:08:28Z","abstract_excerpt":"Visual signals can enhance audiovisual speech recognition accuracy by providing additional contextual information. Given the complexity of visual signals, an audiovisual speech recognition model requires robust generalization capabilities across diverse video scenarios, presenting a significant challenge. In this paper, we introduce EVA, leveraging the mixture-of-Experts for audioVisual ASR to perform robust speech recognition for ``in-the-wild'' videos. Specifically, we first encode visual information into visual tokens sequence and map them into speech space by a lightweight projection. Then"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.12370","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2024-09-19T00:08:28Z","cross_cats_sorted":["cs.CL","cs.CV","cs.SD"],"title_canon_sha256":"e9a83f7e0b87d2ba12f6098c2491cc033cded22ebbdd2180fba7909fea157d17","abstract_canon_sha256":"33a50da710470ca65571297c41eba3c2c303d05fdfc834a21a40f1762d902db2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:08:52.865838Z","signature_b64":"tZyM3apJhlKJb28u7L/ZAg5zXVjWz2IEky1LE3seN7j0Ro6bZvPvCaIZSHVLlSBR2HAnzvZ+J6bDWJSsUsvdCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"09ad6bfbe3b570ce339ee2f502afe404fe28c921eff47aaeb8fa46e7ff212e0c","last_reissued_at":"2026-07-05T09:08:52.865394Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:08:52.865394Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Robust Audiovisual Speech Recognition Models with Mixture-of-Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.SD"],"primary_cat":"eess.AS","authors_text":"Ruihua Song, Shinji Watanabe, Xuankai Chang, Yichen Lu, Yifan Peng, Yihan Wu","submitted_at":"2024-09-19T00:08:28Z","abstract_excerpt":"Visual signals can enhance audiovisual speech recognition accuracy by providing additional contextual information. Given the complexity of visual signals, an audiovisual speech recognition model requires robust generalization capabilities across diverse video scenarios, presenting a significant challenge. In this paper, we introduce EVA, leveraging the mixture-of-Experts for audioVisual ASR to perform robust speech recognition for ``in-the-wild'' videos. Specifically, we first encode visual information into visual tokens sequence and map them into speech space by a lightweight projection. Then"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.12370","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.12370/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.12370","created_at":"2026-07-05T09:08:52.865463+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.12370v1","created_at":"2026-07-05T09:08:52.865463+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.12370","created_at":"2026-07-05T09:08:52.865463+00:00"},{"alias_kind":"pith_short_12","alias_value":"BGWWX67DWVYM","created_at":"2026-07-05T09:08:52.865463+00:00"},{"alias_kind":"pith_short_16","alias_value":"BGWWX67DWVYM4M46","created_at":"2026-07-05T09:08:52.865463+00:00"},{"alias_kind":"pith_short_8","alias_value":"BGWWX67D","created_at":"2026-07-05T09:08:52.865463+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.24059","citing_title":"Towards disentangling the contributions of articulation and acoustics in multimodal phoneme recognition","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BGWWX67DWVYM4M464L2QFL7EAT","json":"https://pith.science/pith/BGWWX67DWVYM4M464L2QFL7EAT.json","graph_json":"https://pith.science/api/pith-number/BGWWX67DWVYM4M464L2QFL7EAT/graph.json","events_json":"https://pith.science/api/pith-number/BGWWX67DWVYM4M464L2QFL7EAT/events.json","paper":"https://pith.science/paper/BGWWX67D"},"agent_actions":{"view_html":"https://pith.science/pith/BGWWX67DWVYM4M464L2QFL7EAT","download_json":"https://pith.science/pith/BGWWX67DWVYM4M464L2QFL7EAT.json","view_paper":"https://pith.science/paper/BGWWX67D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.12370&json=true","fetch_graph":"https://pith.science/api/pith-number/BGWWX67DWVYM4M464L2QFL7EAT/graph.json","fetch_events":"https://pith.science/api/pith-number/BGWWX67DWVYM4M464L2QFL7EAT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BGWWX67DWVYM4M464L2QFL7EAT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BGWWX67DWVYM4M464L2QFL7EAT/action/storage_attestation","attest_author":"https://pith.science/pith/BGWWX67DWVYM4M464L2QFL7EAT/action/author_attestation","sign_citation":"https://pith.science/pith/BGWWX67DWVYM4M464L2QFL7EAT/action/citation_signature","submit_replication":"https://pith.science/pith/BGWWX67DWVYM4M464L2QFL7EAT/action/replication_record"}},"created_at":"2026-07-05T09:08:52.865463+00:00","updated_at":"2026-07-05T09:08:52.865463+00:00"}