{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:W36XQHUPNCBDBHPJIRQH24V6EO","short_pith_number":"pith:W36XQHUP","schema_version":"1.0","canonical_sha256":"b6fd781e8f6882309de944607d72be239fbf69954f9dece0a66f0e6ed95d91b9","source":{"kind":"arxiv","id":"2501.04038","version":1},"attestation_state":"computed","paper":{"title":"Listening and Seeing Again: Generative Error Correction for Audio-Visual Speech Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.MM","authors_text":"Haizhou Li, Hongyu Yuan, Rui Liu","submitted_at":"2025-01-03T10:51:14Z","abstract_excerpt":"Unlike traditional Automatic Speech Recognition (ASR), Audio-Visual Speech Recognition (AVSR) takes audio and visual signals simultaneously to infer the transcription. Recent studies have shown that Large Language Models (LLMs) can be effectively used for Generative Error Correction (GER) in ASR by predicting the best transcription from ASR-generated N-best hypotheses. However, these LLMs lack the ability to simultaneously understand audio and visual, making the GER approach challenging to apply in AVSR. In this work, we propose a novel GER paradigm for AVSR, termed AVGER, that follows the con"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.04038","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.MM","submitted_at":"2025-01-03T10:51:14Z","cross_cats_sorted":["cs.AI","cs.SD","eess.AS"],"title_canon_sha256":"e6d8f0821433eb30d4b6bb2a6e7736619cdd7a607643dc2b34796adb6a5cb104","abstract_canon_sha256":"dc9160e98b766436c5c4be5de067325bf50f36bb0e0d6ea549e8d591ecc6568b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:58:28.838603Z","signature_b64":"mL1/Bt4Z9UTTdWCU47Tiox1ZlJCvE3sDU06Xbrz4+sPgV8ktYhPaNyqwzHWiiHm56EeNzMTphScMxH8DfTpCDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b6fd781e8f6882309de944607d72be239fbf69954f9dece0a66f0e6ed95d91b9","last_reissued_at":"2026-07-05T09:58:28.838187Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:58:28.838187Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Listening and Seeing Again: Generative Error Correction for Audio-Visual Speech Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.MM","authors_text":"Haizhou Li, Hongyu Yuan, Rui Liu","submitted_at":"2025-01-03T10:51:14Z","abstract_excerpt":"Unlike traditional Automatic Speech Recognition (ASR), Audio-Visual Speech Recognition (AVSR) takes audio and visual signals simultaneously to infer the transcription. Recent studies have shown that Large Language Models (LLMs) can be effectively used for Generative Error Correction (GER) in ASR by predicting the best transcription from ASR-generated N-best hypotheses. However, these LLMs lack the ability to simultaneously understand audio and visual, making the GER approach challenging to apply in AVSR. In this work, we propose a novel GER paradigm for AVSR, termed AVGER, that follows the con"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.04038","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.04038/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.04038","created_at":"2026-07-05T09:58:28.838245+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.04038v1","created_at":"2026-07-05T09:58:28.838245+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.04038","created_at":"2026-07-05T09:58:28.838245+00:00"},{"alias_kind":"pith_short_12","alias_value":"W36XQHUPNCBD","created_at":"2026-07-05T09:58:28.838245+00:00"},{"alias_kind":"pith_short_16","alias_value":"W36XQHUPNCBDBHPJ","created_at":"2026-07-05T09:58:28.838245+00:00"},{"alias_kind":"pith_short_8","alias_value":"W36XQHUP","created_at":"2026-07-05T09:58:28.838245+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W36XQHUPNCBDBHPJIRQH24V6EO","json":"https://pith.science/pith/W36XQHUPNCBDBHPJIRQH24V6EO.json","graph_json":"https://pith.science/api/pith-number/W36XQHUPNCBDBHPJIRQH24V6EO/graph.json","events_json":"https://pith.science/api/pith-number/W36XQHUPNCBDBHPJIRQH24V6EO/events.json","paper":"https://pith.science/paper/W36XQHUP"},"agent_actions":{"view_html":"https://pith.science/pith/W36XQHUPNCBDBHPJIRQH24V6EO","download_json":"https://pith.science/pith/W36XQHUPNCBDBHPJIRQH24V6EO.json","view_paper":"https://pith.science/paper/W36XQHUP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.04038&json=true","fetch_graph":"https://pith.science/api/pith-number/W36XQHUPNCBDBHPJIRQH24V6EO/graph.json","fetch_events":"https://pith.science/api/pith-number/W36XQHUPNCBDBHPJIRQH24V6EO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W36XQHUPNCBDBHPJIRQH24V6EO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W36XQHUPNCBDBHPJIRQH24V6EO/action/storage_attestation","attest_author":"https://pith.science/pith/W36XQHUPNCBDBHPJIRQH24V6EO/action/author_attestation","sign_citation":"https://pith.science/pith/W36XQHUPNCBDBHPJIRQH24V6EO/action/citation_signature","submit_replication":"https://pith.science/pith/W36XQHUPNCBDBHPJIRQH24V6EO/action/replication_record"}},"created_at":"2026-07-05T09:58:28.838245+00:00","updated_at":"2026-07-05T09:58:28.838245+00:00"}