{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:C52RJPI4UFTVWX6GSBW7CNCHOS","short_pith_number":"pith:C52RJPI4","schema_version":"1.0","canonical_sha256":"177514bd1ca1675b5fc6906df1344774a4c7c1eaf5315ea1a9120a8bdef2031c","source":{"kind":"arxiv","id":"2506.09792","version":2},"attestation_state":"computed","paper":{"title":"Incorporating Linguistic Constraints from External Knowledge Source for Audio-Visual Target Speech Extraction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Haizhou Li, Helen Meng, Shuai Wang, Wenxuan Wu, Xixin Wu","submitted_at":"2025-06-11T14:36:26Z","abstract_excerpt":"Audio-visual target speaker extraction (AV-TSE) models primarily rely on target visual cues to isolate the target speaker's voice from others. We know that humans leverage linguistic knowledge, such as syntax and semantics, to support speech perception. Inspired by this, we explore the potential of pre-trained speech-language models (PSLMs) and pre-trained language models (PLMs) as auxiliary knowledge sources for AV-TSE. In this study, we propose incorporating the linguistic constraints from PSLMs or PLMs for the AV-TSE model as additional supervision signals. Without introducing any extra com"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.09792","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2025-06-11T14:36:26Z","cross_cats_sorted":["cs.LG","cs.MM","eess.AS"],"title_canon_sha256":"e0435034aced376d327e367f521c737a7f92053b65346f697f4e5364003f8434","abstract_canon_sha256":"a42292bfde37a3d478c346a7c2d30d00c9589d1673e2536ba01d8ddc540792d4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:21:49.452224Z","signature_b64":"WvucUwX5HIJKXkBrG8jH42oQiuyVY3xd+vRLZ94+0fjwi3zatu4V/epue4SvJSrAID4gppmSr4RU5oD1HM82AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"177514bd1ca1675b5fc6906df1344774a4c7c1eaf5315ea1a9120a8bdef2031c","last_reissued_at":"2026-07-05T11:21:49.451773Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:21:49.451773Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Incorporating Linguistic Constraints from External Knowledge Source for Audio-Visual Target Speech Extraction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Haizhou Li, Helen Meng, Shuai Wang, Wenxuan Wu, Xixin Wu","submitted_at":"2025-06-11T14:36:26Z","abstract_excerpt":"Audio-visual target speaker extraction (AV-TSE) models primarily rely on target visual cues to isolate the target speaker's voice from others. We know that humans leverage linguistic knowledge, such as syntax and semantics, to support speech perception. Inspired by this, we explore the potential of pre-trained speech-language models (PSLMs) and pre-trained language models (PLMs) as auxiliary knowledge sources for AV-TSE. In this study, we propose incorporating the linguistic constraints from PSLMs or PLMs for the AV-TSE model as additional supervision signals. Without introducing any extra com"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.09792","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.09792/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.09792","created_at":"2026-07-05T11:21:49.451830+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.09792v2","created_at":"2026-07-05T11:21:49.451830+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.09792","created_at":"2026-07-05T11:21:49.451830+00:00"},{"alias_kind":"pith_short_12","alias_value":"C52RJPI4UFTV","created_at":"2026-07-05T11:21:49.451830+00:00"},{"alias_kind":"pith_short_16","alias_value":"C52RJPI4UFTVWX6G","created_at":"2026-07-05T11:21:49.451830+00:00"},{"alias_kind":"pith_short_8","alias_value":"C52RJPI4","created_at":"2026-07-05T11:21:49.451830+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.09792","citing_title":"Incorporating Linguistic Constraints from External Knowledge Source for Audio-Visual Target Speech Extraction","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C52RJPI4UFTVWX6GSBW7CNCHOS","json":"https://pith.science/pith/C52RJPI4UFTVWX6GSBW7CNCHOS.json","graph_json":"https://pith.science/api/pith-number/C52RJPI4UFTVWX6GSBW7CNCHOS/graph.json","events_json":"https://pith.science/api/pith-number/C52RJPI4UFTVWX6GSBW7CNCHOS/events.json","paper":"https://pith.science/paper/C52RJPI4"},"agent_actions":{"view_html":"https://pith.science/pith/C52RJPI4UFTVWX6GSBW7CNCHOS","download_json":"https://pith.science/pith/C52RJPI4UFTVWX6GSBW7CNCHOS.json","view_paper":"https://pith.science/paper/C52RJPI4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.09792&json=true","fetch_graph":"https://pith.science/api/pith-number/C52RJPI4UFTVWX6GSBW7CNCHOS/graph.json","fetch_events":"https://pith.science/api/pith-number/C52RJPI4UFTVWX6GSBW7CNCHOS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C52RJPI4UFTVWX6GSBW7CNCHOS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C52RJPI4UFTVWX6GSBW7CNCHOS/action/storage_attestation","attest_author":"https://pith.science/pith/C52RJPI4UFTVWX6GSBW7CNCHOS/action/author_attestation","sign_citation":"https://pith.science/pith/C52RJPI4UFTVWX6GSBW7CNCHOS/action/citation_signature","submit_replication":"https://pith.science/pith/C52RJPI4UFTVWX6GSBW7CNCHOS/action/replication_record"}},"created_at":"2026-07-05T11:21:49.451830+00:00","updated_at":"2026-07-05T11:21:49.451830+00:00"}