{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YCGRDQZBUHXUZ7JZQLNM25WG4A","short_pith_number":"pith:YCGRDQZB","schema_version":"1.0","canonical_sha256":"c08d11c321a1ef4cfd3982dacd76c6e036b662e35960e217fd2226a329f8dac9","source":{"kind":"arxiv","id":"2504.07229","version":1},"attestation_state":"computed","paper":{"title":"Visual-Aware Speech Recognition for Noisy Scenarios","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["eess.AS","eess.SP"],"primary_cat":"cs.CL","authors_text":"Karan Singla, Lakshmipathi Balaji","submitted_at":"2025-04-09T19:09:54Z","abstract_excerpt":"Humans have the ability to utilize visual cues, such as lip movements and visual scenes, to enhance auditory perception, particularly in noisy environments. However, current Automatic Speech Recognition (ASR) or Audio-Visual Speech Recognition (AVSR) models often struggle in noisy scenarios. To solve this task, we propose a model that improves transcription by correlating noise sources to visual cues. Unlike works that rely on lip motion and require the speaker's visibility, we exploit broader visual information from the environment. This allows our model to naturally filter speech from noise "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.07229","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-09T19:09:54Z","cross_cats_sorted":["eess.AS","eess.SP"],"title_canon_sha256":"5837057b7b124ab7cf6245952c3358b2b713c216edf9368d88b12d458699487f","abstract_canon_sha256":"e017a747fbb153bc0f926acd2637bf2c837141e92b39ca07ffd084bbf0958f80"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:05.459613Z","signature_b64":"+/rdwJFCMsdPTWUTVqGgTKUW8opn+dpvM29hYk/wklDOUypxdOV3j5La/4SZem2ae0NXrYx0jpEREhvUu77MBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c08d11c321a1ef4cfd3982dacd76c6e036b662e35960e217fd2226a329f8dac9","last_reissued_at":"2026-07-05T10:47:05.459164Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:05.459164Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual-Aware Speech Recognition for Noisy Scenarios","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["eess.AS","eess.SP"],"primary_cat":"cs.CL","authors_text":"Karan Singla, Lakshmipathi Balaji","submitted_at":"2025-04-09T19:09:54Z","abstract_excerpt":"Humans have the ability to utilize visual cues, such as lip movements and visual scenes, to enhance auditory perception, particularly in noisy environments. However, current Automatic Speech Recognition (ASR) or Audio-Visual Speech Recognition (AVSR) models often struggle in noisy scenarios. To solve this task, we propose a model that improves transcription by correlating noise sources to visual cues. Unlike works that rely on lip motion and require the speaker's visibility, we exploit broader visual information from the environment. This allows our model to naturally filter speech from noise "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.07229","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.07229/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.07229","created_at":"2026-07-05T10:47:05.459219+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.07229v1","created_at":"2026-07-05T10:47:05.459219+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.07229","created_at":"2026-07-05T10:47:05.459219+00:00"},{"alias_kind":"pith_short_12","alias_value":"YCGRDQZBUHXU","created_at":"2026-07-05T10:47:05.459219+00:00"},{"alias_kind":"pith_short_16","alias_value":"YCGRDQZBUHXUZ7JZ","created_at":"2026-07-05T10:47:05.459219+00:00"},{"alias_kind":"pith_short_8","alias_value":"YCGRDQZB","created_at":"2026-07-05T10:47:05.459219+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YCGRDQZBUHXUZ7JZQLNM25WG4A","json":"https://pith.science/pith/YCGRDQZBUHXUZ7JZQLNM25WG4A.json","graph_json":"https://pith.science/api/pith-number/YCGRDQZBUHXUZ7JZQLNM25WG4A/graph.json","events_json":"https://pith.science/api/pith-number/YCGRDQZBUHXUZ7JZQLNM25WG4A/events.json","paper":"https://pith.science/paper/YCGRDQZB"},"agent_actions":{"view_html":"https://pith.science/pith/YCGRDQZBUHXUZ7JZQLNM25WG4A","download_json":"https://pith.science/pith/YCGRDQZBUHXUZ7JZQLNM25WG4A.json","view_paper":"https://pith.science/paper/YCGRDQZB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.07229&json=true","fetch_graph":"https://pith.science/api/pith-number/YCGRDQZBUHXUZ7JZQLNM25WG4A/graph.json","fetch_events":"https://pith.science/api/pith-number/YCGRDQZBUHXUZ7JZQLNM25WG4A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YCGRDQZBUHXUZ7JZQLNM25WG4A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YCGRDQZBUHXUZ7JZQLNM25WG4A/action/storage_attestation","attest_author":"https://pith.science/pith/YCGRDQZBUHXUZ7JZQLNM25WG4A/action/author_attestation","sign_citation":"https://pith.science/pith/YCGRDQZBUHXUZ7JZQLNM25WG4A/action/citation_signature","submit_replication":"https://pith.science/pith/YCGRDQZBUHXUZ7JZQLNM25WG4A/action/replication_record"}},"created_at":"2026-07-05T10:47:05.459219+00:00","updated_at":"2026-07-05T10:47:05.459219+00:00"}