{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VLBORAL4JWW35ICB3UEQVSQL4M","short_pith_number":"pith:VLBORAL4","schema_version":"1.0","canonical_sha256":"aac2e8817c4dadbea041dd090aca0be32a5ad3f1fa232c3477bc94f49d2ef24b","source":{"kind":"arxiv","id":"2507.09510","version":3},"attestation_state":"computed","paper":{"title":"Enhancing Target Speaker Extraction with Explicit Speaker Consistency Modeling","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Anbin QI, Shu Wu, Xiang Xie, Yanzhang Xie","submitted_at":"2025-07-13T06:35:13Z","abstract_excerpt":"Target Speaker Extraction (TSE) uses a reference cue to extract the target speech from a mixture. In TSE systems relying on audio cues, the speaker embedding from the enrolled speech is crucial to performance. However, these embeddings may suffer from speaker identity confusion. Unlike previous studies that focus on improving speaker embedding extraction, we improve TSE performance from the perspective of speaker consistency. In this paper, we propose a speaker consistency-aware target speaker extraction method that incorporates a centroid-based speaker consistency loss. This approach enhances"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.09510","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.SD","submitted_at":"2025-07-13T06:35:13Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"252b7023b1b70137b4d9969030475dfdcaa901157b2807135bcebe8abba34b3c","abstract_canon_sha256":"378d18344e27bccb63b1ca4a26a49af51bb18c5313fd586189bfae55e661d61e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:51:31.984629Z","signature_b64":"OXp6UwNW4WaIhlet74CAfVWwEF0inCEIh8c9wHMC+5cq+3b+mE5khedC3oxLT7VUZDS9SHf4UlJRxzvZsSqWDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aac2e8817c4dadbea041dd090aca0be32a5ad3f1fa232c3477bc94f49d2ef24b","last_reissued_at":"2026-07-05T11:51:31.984121Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:51:31.984121Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enhancing Target Speaker Extraction with Explicit Speaker Consistency Modeling","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Anbin QI, Shu Wu, Xiang Xie, Yanzhang Xie","submitted_at":"2025-07-13T06:35:13Z","abstract_excerpt":"Target Speaker Extraction (TSE) uses a reference cue to extract the target speech from a mixture. In TSE systems relying on audio cues, the speaker embedding from the enrolled speech is crucial to performance. However, these embeddings may suffer from speaker identity confusion. Unlike previous studies that focus on improving speaker embedding extraction, we improve TSE performance from the perspective of speaker consistency. In this paper, we propose a speaker consistency-aware target speaker extraction method that incorporates a centroid-based speaker consistency loss. This approach enhances"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.09510","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.09510/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.09510","created_at":"2026-07-05T11:51:31.984179+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.09510v3","created_at":"2026-07-05T11:51:31.984179+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.09510","created_at":"2026-07-05T11:51:31.984179+00:00"},{"alias_kind":"pith_short_12","alias_value":"VLBORAL4JWW3","created_at":"2026-07-05T11:51:31.984179+00:00"},{"alias_kind":"pith_short_16","alias_value":"VLBORAL4JWW35ICB","created_at":"2026-07-05T11:51:31.984179+00:00"},{"alias_kind":"pith_short_8","alias_value":"VLBORAL4","created_at":"2026-07-05T11:51:31.984179+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.09510","citing_title":"Enhancing Target Speaker Extraction with Explicit Speaker Consistency Modeling","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VLBORAL4JWW35ICB3UEQVSQL4M","json":"https://pith.science/pith/VLBORAL4JWW35ICB3UEQVSQL4M.json","graph_json":"https://pith.science/api/pith-number/VLBORAL4JWW35ICB3UEQVSQL4M/graph.json","events_json":"https://pith.science/api/pith-number/VLBORAL4JWW35ICB3UEQVSQL4M/events.json","paper":"https://pith.science/paper/VLBORAL4"},"agent_actions":{"view_html":"https://pith.science/pith/VLBORAL4JWW35ICB3UEQVSQL4M","download_json":"https://pith.science/pith/VLBORAL4JWW35ICB3UEQVSQL4M.json","view_paper":"https://pith.science/paper/VLBORAL4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.09510&json=true","fetch_graph":"https://pith.science/api/pith-number/VLBORAL4JWW35ICB3UEQVSQL4M/graph.json","fetch_events":"https://pith.science/api/pith-number/VLBORAL4JWW35ICB3UEQVSQL4M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VLBORAL4JWW35ICB3UEQVSQL4M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VLBORAL4JWW35ICB3UEQVSQL4M/action/storage_attestation","attest_author":"https://pith.science/pith/VLBORAL4JWW35ICB3UEQVSQL4M/action/author_attestation","sign_citation":"https://pith.science/pith/VLBORAL4JWW35ICB3UEQVSQL4M/action/citation_signature","submit_replication":"https://pith.science/pith/VLBORAL4JWW35ICB3UEQVSQL4M/action/replication_record"}},"created_at":"2026-07-05T11:51:31.984179+00:00","updated_at":"2026-07-05T11:51:31.984179+00:00"}