{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:EFJDPJGWO4GWZ2TVYBU3OSHVKV","short_pith_number":"pith:EFJDPJGW","schema_version":"1.0","canonical_sha256":"215237a4d6770d6cea75c069b748f55561c35e85fd4f1eb6c2d68f17a2a4817d","source":{"kind":"arxiv","id":"2607.09001","version":1},"attestation_state":"computed","paper":{"title":"Optimal Transport-based Semantic Alignment for LLM-based Audio-Visual Speech Recognition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Hisashi Kawai, Peng Shen, Xugang Lu, Yu Tsao","submitted_at":"2026-07-10T00:07:25Z","abstract_excerpt":"Large language model (LLM)-based audio-visual speech recognition (LLM-AVSR) has recently demonstrated strong robustness in adverse acoustic environments by leveraging complementary audio and visual information. Existing approaches typically employ independently pretrained acoustic and visual encoders, whose outputs are projected and fused as soft prompts to condition an LLM for speech recognition. However, most methods perform multimodal fusion without explicitly addressing the representational discrepancy between audio, visual and text modalities, potentially limiting the effectiveness of cro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.09001","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2026-07-10T00:07:25Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"2b3ffb87c1fc5ffde364236bebc1c1bd339b4663daa5e8d6a365920939ba87a1","abstract_canon_sha256":"5d23b4543988d53788d61d1e55af49c5ea8cd26198ac90a89eb22568c714fed4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-13T00:17:33.388879Z","signature_b64":"IBThgn0+79L/F9RpOu6Fd7R9McQXxfmDbvMMtePIt6Zn0qrTAK+Z2Vz3V4hjFJaj9tdzAY14bQih5uKdhQX2CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"215237a4d6770d6cea75c069b748f55561c35e85fd4f1eb6c2d68f17a2a4817d","last_reissued_at":"2026-07-13T00:17:33.387859Z","signature_status":"signed_v1","first_computed_at":"2026-07-13T00:17:33.387859Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Optimal Transport-based Semantic Alignment for LLM-based Audio-Visual Speech Recognition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Hisashi Kawai, Peng Shen, Xugang Lu, Yu Tsao","submitted_at":"2026-07-10T00:07:25Z","abstract_excerpt":"Large language model (LLM)-based audio-visual speech recognition (LLM-AVSR) has recently demonstrated strong robustness in adverse acoustic environments by leveraging complementary audio and visual information. Existing approaches typically employ independently pretrained acoustic and visual encoders, whose outputs are projected and fused as soft prompts to condition an LLM for speech recognition. However, most methods perform multimodal fusion without explicitly addressing the representational discrepancy between audio, visual and text modalities, potentially limiting the effectiveness of cro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.09001","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.09001/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.09001","created_at":"2026-07-13T00:17:33.388399+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.09001v1","created_at":"2026-07-13T00:17:33.388399+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.09001","created_at":"2026-07-13T00:17:33.388399+00:00"},{"alias_kind":"pith_short_12","alias_value":"EFJDPJGWO4GW","created_at":"2026-07-13T00:17:33.388399+00:00"},{"alias_kind":"pith_short_16","alias_value":"EFJDPJGWO4GWZ2TV","created_at":"2026-07-13T00:17:33.388399+00:00"},{"alias_kind":"pith_short_8","alias_value":"EFJDPJGW","created_at":"2026-07-13T00:17:33.388399+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EFJDPJGWO4GWZ2TVYBU3OSHVKV","json":"https://pith.science/pith/EFJDPJGWO4GWZ2TVYBU3OSHVKV.json","graph_json":"https://pith.science/api/pith-number/EFJDPJGWO4GWZ2TVYBU3OSHVKV/graph.json","events_json":"https://pith.science/api/pith-number/EFJDPJGWO4GWZ2TVYBU3OSHVKV/events.json","paper":"https://pith.science/paper/EFJDPJGW"},"agent_actions":{"view_html":"https://pith.science/pith/EFJDPJGWO4GWZ2TVYBU3OSHVKV","download_json":"https://pith.science/pith/EFJDPJGWO4GWZ2TVYBU3OSHVKV.json","view_paper":"https://pith.science/paper/EFJDPJGW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.09001&json=true","fetch_graph":"https://pith.science/api/pith-number/EFJDPJGWO4GWZ2TVYBU3OSHVKV/graph.json","fetch_events":"https://pith.science/api/pith-number/EFJDPJGWO4GWZ2TVYBU3OSHVKV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EFJDPJGWO4GWZ2TVYBU3OSHVKV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EFJDPJGWO4GWZ2TVYBU3OSHVKV/action/storage_attestation","attest_author":"https://pith.science/pith/EFJDPJGWO4GWZ2TVYBU3OSHVKV/action/author_attestation","sign_citation":"https://pith.science/pith/EFJDPJGWO4GWZ2TVYBU3OSHVKV/action/citation_signature","submit_replication":"https://pith.science/pith/EFJDPJGWO4GWZ2TVYBU3OSHVKV/action/replication_record"}},"created_at":"2026-07-13T00:17:33.388399+00:00","updated_at":"2026-07-13T00:17:33.388399+00:00"}