{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:L37XBUY6OBF7WXBYOIZHD5T7RM","short_pith_number":"pith:L37XBUY6","schema_version":"1.0","canonical_sha256":"5eff70d31e704bfb5c38723271f67f8b349835c9b0004b402d63830f5bc6555c","source":{"kind":"arxiv","id":"1904.07845","version":1},"attestation_state":"computed","paper":{"title":"Improved Speech Separation with Time-and-Frequency Cross-domain Joint Embedding and Clustering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS","stat.ML"],"primary_cat":"cs.SD","authors_text":"Chao-I Tuan, Gene-Ping Yang, Hung-yi Lee, Lin-shan Lee","submitted_at":"2019-04-16T17:48:59Z","abstract_excerpt":"Speech separation has been very successful with deep learning techniques. Substantial effort has been reported based on approaches over spectrogram, which is well known as the standard time-and-frequency cross-domain representation for speech signals. It is highly correlated to the phonetic structure of speech, or \"how the speech sounds\" when perceived by human, but primarily frequency domain features carrying temporal behaviour. Very impressive work achieving speech separation over time domain was reported recently, probably because waveforms in time domain may describe the different realizat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1904.07845","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2019-04-16T17:48:59Z","cross_cats_sorted":["cs.LG","eess.AS","stat.ML"],"title_canon_sha256":"2d38f0c21373e3f31330a0747632d22f5b711d446c07e099b1acea4ff1b26649","abstract_canon_sha256":"72c142606950a4a8488a53aa1a3591c26fb47dca2adfcfcc20888a740158d40c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-17T23:48:24.069780Z","signature_b64":"hPjboTpfT47mH2YuR/FqLVb0/w8yKLWdrf6AVYwFMeyOav7lly7FG06+Bd6L4b/GLbAQMDV1X0bZ2Ekr3WgsAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5eff70d31e704bfb5c38723271f67f8b349835c9b0004b402d63830f5bc6555c","last_reissued_at":"2026-05-17T23:48:24.069245Z","signature_status":"signed_v1","first_computed_at":"2026-05-17T23:48:24.069245Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improved Speech Separation with Time-and-Frequency Cross-domain Joint Embedding and Clustering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS","stat.ML"],"primary_cat":"cs.SD","authors_text":"Chao-I Tuan, Gene-Ping Yang, Hung-yi Lee, Lin-shan Lee","submitted_at":"2019-04-16T17:48:59Z","abstract_excerpt":"Speech separation has been very successful with deep learning techniques. Substantial effort has been reported based on approaches over spectrogram, which is well known as the standard time-and-frequency cross-domain representation for speech signals. It is highly correlated to the phonetic structure of speech, or \"how the speech sounds\" when perceived by human, but primarily frequency domain features carrying temporal behaviour. Very impressive work achieving speech separation over time domain was reported recently, probably because waveforms in time domain may describe the different realizat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1904.07845","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1904.07845","created_at":"2026-05-17T23:48:24.069328+00:00"},{"alias_kind":"arxiv_version","alias_value":"1904.07845v1","created_at":"2026-05-17T23:48:24.069328+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1904.07845","created_at":"2026-05-17T23:48:24.069328+00:00"},{"alias_kind":"pith_short_12","alias_value":"L37XBUY6OBF7","created_at":"2026-05-18T12:33:21.387695+00:00"},{"alias_kind":"pith_short_16","alias_value":"L37XBUY6OBF7WXBY","created_at":"2026-05-18T12:33:21.387695+00:00"},{"alias_kind":"pith_short_8","alias_value":"L37XBUY6","created_at":"2026-05-18T12:33:21.387695+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.08375","citing_title":"Developing an Effective Training Dataset to Enhance the Performance of AI-based Speaker Separation Systems","ref_index":5,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L37XBUY6OBF7WXBYOIZHD5T7RM","json":"https://pith.science/pith/L37XBUY6OBF7WXBYOIZHD5T7RM.json","graph_json":"https://pith.science/api/pith-number/L37XBUY6OBF7WXBYOIZHD5T7RM/graph.json","events_json":"https://pith.science/api/pith-number/L37XBUY6OBF7WXBYOIZHD5T7RM/events.json","paper":"https://pith.science/paper/L37XBUY6"},"agent_actions":{"view_html":"https://pith.science/pith/L37XBUY6OBF7WXBYOIZHD5T7RM","download_json":"https://pith.science/pith/L37XBUY6OBF7WXBYOIZHD5T7RM.json","view_paper":"https://pith.science/paper/L37XBUY6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1904.07845&json=true","fetch_graph":"https://pith.science/api/pith-number/L37XBUY6OBF7WXBYOIZHD5T7RM/graph.json","fetch_events":"https://pith.science/api/pith-number/L37XBUY6OBF7WXBYOIZHD5T7RM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L37XBUY6OBF7WXBYOIZHD5T7RM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L37XBUY6OBF7WXBYOIZHD5T7RM/action/storage_attestation","attest_author":"https://pith.science/pith/L37XBUY6OBF7WXBYOIZHD5T7RM/action/author_attestation","sign_citation":"https://pith.science/pith/L37XBUY6OBF7WXBYOIZHD5T7RM/action/citation_signature","submit_replication":"https://pith.science/pith/L37XBUY6OBF7WXBYOIZHD5T7RM/action/replication_record"}},"created_at":"2026-05-17T23:48:24.069328+00:00","updated_at":"2026-05-17T23:48:24.069328+00:00"}