{"as_of":"2026-08-13T06:33:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:69fc8b805b433251195eb86bc071422b8f69281ebd7e95a980684eb8a5b5abbb","coverage":[{"denominator":25,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":25,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T15:53:11.146098Z","state":"measured"},{"denominator":25,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":25,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2411.13811/citation-record","integrity":"/paper/2411.13811/integrity","json":"/paper/2411.13811/citation-record.json","paper":"/paper/2411.13811"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.477923Z","title":"Deep clustering: Discriminative embeddings for segmentation and separation,","venue":null,"work_id":"c5b4708e-1f8b-486d-888a-6802224ddea8","year":2016},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.024260Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:2ca49fdb6f014b37b788f41f46162ea936087571fa2ea5c44446b95d8f740dee","observation_id":"f7a604b1-96b1-4575-8948-c2f63433021d","resolution":{"observed_at":"2026-08-12T15:53:11.482346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.12766","last_updated":"2020-10-24T03:57:19Z","snapshot_observed_at":"2026-08-06T11:01:30.886477Z","submitted_at":"2020-10-24T03:57:19Z","title":"X-TaSNet: Robust and Accurate Time-Domain Speaker Extraction Network","version":1},"cited_work":{"arxiv_id":"2010.12766","doi":null,"metadata_source":"pith","pith_arxiv_id":"2010.12766","snapshot_observed_at":"2026-08-12T15:53:11.290499Z","title":"X-TaSNet: Robust and Accurate Time-Domain Speaker Extraction Network","venue":"eess.AS","work_id":"dd65dcc7-dbb0-458a-b285-718623cb2b63","year":2020},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.029868Z"},"links":{"cited_paper":"/paper/2010.12766","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:0f6dc1c7c90b61b6fa00acf4666e56474d3c7533173afeb12ce247cac749fddb","observation_id":"04028720-58e2-4d16-9c4d-adc632d03d8a","resolution":{"observed_at":"2026-08-12T15:53:11.297492Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.035688Z","title":"Conv-tasnet: Surpassing ideal time– frequency magnitude masking for speech separation,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.035688Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:e6cd1a8ad030f6a89d1a632394c54e3e0c4c1fb4f4d5a8606babbce07a480e16","observation_id":"e79aa124-4c5b-4036-a234-e0a6391e6eb4","resolution":{"observed_at":"2026-08-12T15:53:11.035688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.459287Z","title":"X-sepformer: End-to-end speaker extraction network with explicit optimization on speaker confusion,","venue":null,"work_id":"791f6148-6321-457a-84e2-dd8f230dbadb","year":2023},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.040021Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:19bca639982b60f1512080e33ea7a97d04771713281ba5de3e5b531e383daf9c","observation_id":"22ae94b1-0ec2-49a6-82f5-d78522e1e266","resolution":{"observed_at":"2026-08-12T15:53:11.463604Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.044480Z","title":"Attention is all you need in speech separation,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.044480Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:30b5b442beb95d98308838c7f1ec364db842d30abcb2f9c562837913821c76ac","observation_id":"6ab5cef4-d449-410e-9d3a-646586da20d4","resolution":{"observed_at":"2026-08-12T15:53:11.044480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.441688Z","title":"X-tf-gridnet: A time–frequency domain target speaker extraction network with adaptive speaker embedding fusion,","venue":null,"work_id":"58f4fa29-5610-4bb2-acb0-c37ca4cee67f","year":2024},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.048595Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:2126cb7bd59e61867f8ae65ffe0b25330d88c5b90e54d458dc5533ba753e3b05","observation_id":"b5b27d92-39b9-4bb9-962d-0d7982f12f4c","resolution":{"observed_at":"2026-08-12T15:53:11.445701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.430512Z","title":"Tf-gridnet: Making time-frequency domain models great again for monaural speaker separation,","venue":null,"work_id":"4b887623-5369-4d21-a723-093f1a521591","year":2023},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.053272Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:06937f4e3fea48615429fa3dcb006c71ac588889c2523b95aeaf17c8b1b56d2a","observation_id":"71ff4977-d239-43e8-9f4d-f8fb153437bc","resolution":{"observed_at":"2026-08-12T15:53:11.434340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.419189Z","title":"Single channel target speaker extraction and recognition with speaker beam,","venue":null,"work_id":"3c8228bc-2726-4c4c-92c9-ef98df24ad4e","year":2018},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.057050Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:05891cd42992628649f2b19776bacdf39300b03d6c5c9b9d0ad8f8edeef02905","observation_id":"b993366f-2619-4d54-b3dc-91e0af823588","resolution":{"observed_at":"2026-08-12T15:53:11.423397Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.04826","last_updated":"2019-06-19T17:10:51Z","snapshot_observed_at":"2026-08-12T02:45:02.906532Z","submitted_at":"2018-10-11T02:57:14Z","title":"VoiceFilter: Targeted Voice Separation by Speaker-Conditioned Spectrogram Masking","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.04826","snapshot_observed_at":"2026-08-12T15:53:11.061029Z","title":"V oicefilter: Targeted voice separation by speaker-conditioned spectrogram masking,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.061029Z"},"links":{"cited_paper":"/paper/1810.04826","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:79c42026342209d37440963ddc5fd661a97b492d5a4844306cfeda99e39167d0","observation_id":"8841318e-8b8e-458b-876c-4e54099e768f","resolution":{"observed_at":"2026-08-12T15:53:11.061029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2202.00733","last_updated":"2023-09-15T06:15:22Z","snapshot_observed_at":"2026-08-04T02:51:19.469927Z","submitted_at":"2022-02-01T20:10:23Z","title":"New Insights on Target Speaker Extraction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.00733","snapshot_observed_at":"2026-08-12T15:53:11.066619Z","title":"New insights on target speaker extraction,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.066619Z"},"links":{"cited_paper":"/paper/2202.00733","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:623c755be4252fdb75691f44271e28dfe3a01a0fba7c1c70e39f4638088973fb","observation_id":"27de0348-9ded-4fb5-97f8-4aec62bab1ba","resolution":{"observed_at":"2026-08-12T15:53:11.066619Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.071241Z","title":"Wavesplit: End-to-end speech separation by speaker clustering,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.071241Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:1182b7e05cddda7de4e7ae7002ef462796d99572a6ebb16fb83fc2c0c596e2bf","observation_id":"c6689993-6469-480e-9fe0-dcd336a4d983","resolution":{"observed_at":"2026-08-12T15:53:11.071241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.399948Z","title":"Attention-based scaling adaptation for target speech extraction,","venue":null,"work_id":"205344cf-fb42-45b3-a054-7fcb7f97c90c","year":2021},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.075799Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:487cf516dca8d4c6b8a025a14f6dc90a0058315619525ee49843c45641141d82","observation_id":"4ac044fc-519e-4126-9605-4f399e024254","resolution":{"observed_at":"2026-08-12T15:53:11.404027Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.386784Z","title":"Target speaker extraction by directly exploiting contextual information in the time-frequency domain,","venue":null,"work_id":"88e35314-6697-48f4-8e4e-03bc11c14025","year":2024},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.083679Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:34861ffd9bef552549f1270fc857cec4e5a70dc082ea4a4bcc77d8aefbdfa410","observation_id":"a3cee24f-c101-48da-b1e6-300af3a912b7","resolution":{"observed_at":"2026-08-12T15:53:11.392409Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03411","last_updated":"2024-03-06T02:39:21Z","snapshot_observed_at":"2026-08-13T04:02:35.088214Z","submitted_at":"2024-03-06T02:39:21Z","title":"CrossNet: Leveraging Global, Cross-Band, Narrow-Band, and Positional Encoding for Single- and Multi-Channel Speaker Separation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03411","snapshot_observed_at":"2026-08-12T15:53:11.087244Z","title":"Crossnet: Leveraging global, cross- band, narrow-band, and positional encoding for single-and multi-channel speaker separation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.087244Z"},"links":{"cited_paper":"/paper/2403.03411","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:56fa6cce3b20eaba47c6fba2c0e0ab912fc4163f26bd00350332994cb1b6a47a","observation_id":"2e8fd2de-7ed2-433d-9b4c-5ca83e305eee","resolution":{"observed_at":"2026-08-12T15:53:11.087244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.08100","last_updated":"2020-05-16T20:56:25Z","snapshot_observed_at":"2026-08-08T15:18:32.322111Z","submitted_at":"2020-05-16T20:56:25Z","title":"Conformer: Convolution-augmented Transformer for Speech Recognition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.08100","snapshot_observed_at":"2026-08-12T15:53:11.092180Z","title":"Conformer: Convolution-augmented transformer for speech recognition,","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.092180Z"},"links":{"cited_paper":"/paper/2005.08100","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:4bf479e33c61066ea445d249f426798458fe6077c76947f749173e215236b43d","observation_id":"4f747a62-0642-4d58-9448-abbc1ee71cb4","resolution":{"observed_at":"2026-08-12T15:53:11.092180Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.095809Z","title":"Sdr–half-baked or well done?","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.095809Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:14f0c556583b7c7433112ab49375af942e557464837e01fa9899710565d2f18e","observation_id":"ce92acf1-fd2f-4e0d-ab46-db00a288aadd","resolution":{"observed_at":"2026-08-12T15:53:11.095809Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04686","last_updated":"2020-08-18T03:03:01Z","snapshot_observed_at":"2026-08-12T03:39:08.946505Z","submitted_at":"2020-05-10T15:00:07Z","title":"SpEx+: A Complete Time Domain Speaker Extraction Network","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.04686","snapshot_observed_at":"2026-08-12T15:53:11.099817Z","title":"Spex+: A complete time domain speaker extraction network,","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.099817Z"},"links":{"cited_paper":"/paper/2005.04686","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:7caf5f0b723a1520f68ee84463080d51c602c57c381f2db0ec6cb88d3369c6e8","observation_id":"5a0d7351-89b4-4407-8c88-f4994ad21691","resolution":{"observed_at":"2026-08-12T15:53:11.099817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.364489Z","title":"Whamr!: Noisy and reverberant single-channel speech separation,","venue":null,"work_id":"22dd03e4-6de1-4d38-95dd-d6f56cf7d84f","year":2020},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.106726Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:afcb1edcb3fd841903cab4fb11d9892fed86d5d01d55611db75d18da35ec890c","observation_id":"1b66e01e-0d20-4204-9868-480abdfaf3b2","resolution":{"observed_at":"2026-08-12T15:53:11.369502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.01160","last_updated":"2019-07-02T04:27:55Z","snapshot_observed_at":"2026-07-06T08:04:17.909965Z","submitted_at":"2019-07-02T04:27:55Z","title":"WHAM!: Extending Speech Separation to Noisy Environments","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1907.01160","snapshot_observed_at":"2026-08-12T15:53:11.110936Z","title":"Wham!: Extending speech separation to noisy environments,","venue":null,"work_id":null,"year":1907},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.110936Z"},"links":{"cited_paper":"/paper/1907.01160","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:31f44ad873f01864fc37b6d2b3e8f0fa38a09839558d0c325e5184188b9bd240","observation_id":"12820096-87ab-49d4-b27a-b51b73842a0e","resolution":{"observed_at":"2026-08-12T15:53:11.110936Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.350577Z","title":"Time-domain speaker extraction network,","venue":null,"work_id":"319f3e9c-8d8e-426d-8c11-da011423a974","year":2019},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.119063Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:3b3cc2d3be6e58f3e392e3e1af90741ef961ca5dc86fc54fad02f3b48c2d5f8f","observation_id":"666982a8-6f6a-487e-beb6-e6c9e047c42f","resolution":{"observed_at":"2026-08-12T15:53:11.354880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.337058Z","title":"Spex: Multi-scale time domain speaker extraction network,","venue":null,"work_id":"f25dfe5c-a22c-4e7f-899e-e3af7387f730","year":2020},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.124767Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:ed29a0680ec4b36f85b45cd32eeb5ba6e9a8e8d196232d59538b252004128294","observation_id":"1a39e866-ee0b-415b-80e4-a2c74f7f8a0f","resolution":{"observed_at":"2026-08-12T15:53:11.341213Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.325156Z","title":"Sef-net: Speaker embedding free target speaker extraction network,","venue":null,"work_id":"82d8dc39-adba-4f6d-92df-71469b30a294","year":2023},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.128516Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:bab3361f543b14acb4cf4666a072c567f88ee78981d0059c073c5fcb2e6233f5","observation_id":"e855fa65-3bdb-4a7a-b8a3-aea102bad867","resolution":{"observed_at":"2026-08-12T15:53:11.329226Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.134385Z","title":"Spatialnet: Extensively learning spatial information for multichannel joint speech separation, denoising and dereverberation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.134385Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:4b5f86a01c9522616aca161b5f07f67abb0061b12f61223dc1fbc6f86eebd9c9","observation_id":"610054a0-a670-464d-bde2-d4c5d665fedc","resolution":{"observed_at":"2026-08-12T15:53:11.134385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-08-09T20:34:52.923500Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-12T15:53:11.139423Z","title":"Decoupled weight decay regularization,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.139423Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:a2eeea9bef210e6d0c4f1c5964b97dda87b949ee64df7eec409f8e3abe0747bf","observation_id":"e8685796-2d3d-47df-bf46-211787792bf9","resolution":{"observed_at":"2026-08-12T15:53:11.139423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.305298Z","title":"Performance measurement in blind audio source separation,","venue":null,"work_id":"af112a6a-e922-425d-b3b3-9403f7b5bf3b","year":2006},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.146098Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:1ad2a71f11ce60e8adb265b409b4767532e8df4949ba54cc72ada70e8064f62e","observation_id":"286d71a3-0744-4ae6-95ab-d92c2b8aa724","resolution":{"observed_at":"2026-08-12T15:53:11.308970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","latest_version":2,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-12T15:48:56.756328Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion"},"reference_resolution":{"displayed":25,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":12,"verified_exact":1,"verified_fuzzy":12},"total_outbound_references":25},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 25 of 25 outbound references and 0 inbound Pith citation observations for arXiv:2411.13811."}