{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5767CA7M4Y67QVIGWSMNVFINVK","short_pith_number":"pith:5767CA7M","schema_version":"1.0","canonical_sha256":"effdf103ece63df85506b498da950daaa7804c371674fe081fa3041380f623c0","source":{"kind":"arxiv","id":"2506.01483","version":3},"attestation_state":"computed","paper":{"title":"Inter-Speaker Relative Cues for Text-Guided Target Speech Extraction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Archontis Politis, Tuomas Virtanen, Wang Dai","submitted_at":"2025-06-02T09:43:43Z","abstract_excerpt":"We propose a novel approach that utilizes inter-speaker relative cues to distinguish target speakers and extract their voices from mixtures. Continuous cues (e.g., temporal order, age, pitch level) are grouped by relative differences, while discrete cues (e.g., language, gender, emotion) retain their categorical distinctions. Compared to fixed speech attribute classification, inter-speaker relative cues offer greater flexibility, facilitating much easier expansion of text-guided target speech extraction datasets. Our experiments show that combining all relative cues yields better performance t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.01483","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2025-06-02T09:43:43Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"69001ba94cb3bedf3f31a44fa655c1b96fc5c2b7ec4778796984f4b1ee44e1f2","abstract_canon_sha256":"61d998bd394f88dac47569a67c1bca5ca2835700e37d5b2573c381520ffc6428"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:17:46.742393Z","signature_b64":"OLOEGyb6nVwZQmV0ddSM+uqyaodAv5Rw352lLW7aoS73u8Aza66bdkhEcEwK/tGlkxVV1RkVQ9aZyjtR7wxpDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"effdf103ece63df85506b498da950daaa7804c371674fe081fa3041380f623c0","last_reissued_at":"2026-07-05T11:17:46.741955Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:17:46.741955Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Inter-Speaker Relative Cues for Text-Guided Target Speech Extraction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Archontis Politis, Tuomas Virtanen, Wang Dai","submitted_at":"2025-06-02T09:43:43Z","abstract_excerpt":"We propose a novel approach that utilizes inter-speaker relative cues to distinguish target speakers and extract their voices from mixtures. Continuous cues (e.g., temporal order, age, pitch level) are grouped by relative differences, while discrete cues (e.g., language, gender, emotion) retain their categorical distinctions. Compared to fixed speech attribute classification, inter-speaker relative cues offer greater flexibility, facilitating much easier expansion of text-guided target speech extraction datasets. Our experiments show that combining all relative cues yields better performance t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.01483","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.01483/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.01483","created_at":"2026-07-05T11:17:46.742009+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.01483v3","created_at":"2026-07-05T11:17:46.742009+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.01483","created_at":"2026-07-05T11:17:46.742009+00:00"},{"alias_kind":"pith_short_12","alias_value":"5767CA7M4Y67","created_at":"2026-07-05T11:17:46.742009+00:00"},{"alias_kind":"pith_short_16","alias_value":"5767CA7M4Y67QVIG","created_at":"2026-07-05T11:17:46.742009+00:00"},{"alias_kind":"pith_short_8","alias_value":"5767CA7M","created_at":"2026-07-05T11:17:46.742009+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.01483","citing_title":"Inter-Speaker Relative Cues for Text-Guided Target Speech Extraction","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5767CA7M4Y67QVIGWSMNVFINVK","json":"https://pith.science/pith/5767CA7M4Y67QVIGWSMNVFINVK.json","graph_json":"https://pith.science/api/pith-number/5767CA7M4Y67QVIGWSMNVFINVK/graph.json","events_json":"https://pith.science/api/pith-number/5767CA7M4Y67QVIGWSMNVFINVK/events.json","paper":"https://pith.science/paper/5767CA7M"},"agent_actions":{"view_html":"https://pith.science/pith/5767CA7M4Y67QVIGWSMNVFINVK","download_json":"https://pith.science/pith/5767CA7M4Y67QVIGWSMNVFINVK.json","view_paper":"https://pith.science/paper/5767CA7M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.01483&json=true","fetch_graph":"https://pith.science/api/pith-number/5767CA7M4Y67QVIGWSMNVFINVK/graph.json","fetch_events":"https://pith.science/api/pith-number/5767CA7M4Y67QVIGWSMNVFINVK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5767CA7M4Y67QVIGWSMNVFINVK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5767CA7M4Y67QVIGWSMNVFINVK/action/storage_attestation","attest_author":"https://pith.science/pith/5767CA7M4Y67QVIGWSMNVFINVK/action/author_attestation","sign_citation":"https://pith.science/pith/5767CA7M4Y67QVIGWSMNVFINVK/action/citation_signature","submit_replication":"https://pith.science/pith/5767CA7M4Y67QVIGWSMNVFINVK/action/replication_record"}},"created_at":"2026-07-05T11:17:46.742009+00:00","updated_at":"2026-07-05T11:17:46.742009+00:00"}