{"as_of":"2026-08-22T08:47:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:03225afefb998d95e34c091200afb9c48e947375630aa2e12df8847032e61da7","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T00:52:00.506822Z","state":"measured"},{"denominator":38,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":38,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.19248/citation-record","integrity":"/paper/2412.19248/integrity","json":"/paper/2412.19248/citation-record.json","paper":"/paper/2412.19248"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.312669Z","title":"Speech enhancement based on deep denoising autoencoder,","venue":null,"work_id":"34e4429a-5209-4b0d-bcbe-c227a00e9a1b","year":2013},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.268437Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:150c655b05d7e02f065572235de74da54ddf23380762bdbdffaa97fe556a9c04","observation_id":"c0338c91-d433-4431-8085-492a15e1b541","resolution":{"observed_at":"2026-08-11T00:52:01.317923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.293048Z","title":"An experimental study on speech enhancement based on deep neural networks,","venue":null,"work_id":"9de1ccf7-4db3-4049-b87c-cbc0fa0b818b","year":2013},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.279362Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:fcee3cefe9ee4ca799966167f9692b80f5bb62808f81266d54e794403a34ccc8","observation_id":"5bbe123e-f4c9-4b6e-899b-4c8d6e35c7c5","resolution":{"observed_at":"2026-08-11T00:52:01.300214Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.272495Z","title":"Multiple-target deep learning for LSTM- RNN based speech enhancement,","venue":null,"work_id":"a21771c3-7188-4e48-abac-123539dc727f","year":2017},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.286053Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:8925b2331a2b7dde877278d2bd2e9996829ebe4388361701b1858356020b0bba","observation_id":"78b534da-3370-40ba-9434-3154e846fd68","resolution":{"observed_at":"2026-08-11T00:52:01.279307Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.226978Z","title":"Phase-aware speech enhancement with deep complex U-net,","venue":null,"work_id":"41e2ed82-426a-4b7f-b02e-5d7fdd2f9d05","year":2018},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.299736Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:162618e7788f5ec68f278a75cbf9cb4e987b834b23633f29c7ae4fb3bf6f3bec","observation_id":"6bf00319-956b-4f9d-9590-9704d915e1a3","resolution":{"observed_at":"2026-08-11T00:52:01.233166Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.207763Z","title":"U-net: Convolu - tional networks for biomedical image segmentation,","venue":null,"work_id":"ab53424d-94df-42da-bbfe-99b4aa3dd20e","year":2015},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.307559Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:9881de11da6a7afbd03188b3fda318d9c0502722c5b2c9585028c5ce0e13a09c","observation_id":"623d7b29-92c5-4d16-8562-404e398360f5","resolution":{"observed_at":"2026-08-11T00:52:01.214091Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.188334Z","title":"DCCRN: Deep complex convolution recurrent network for phase-aware speech enhancement,","venue":null,"work_id":"e7d150bf-5990-4054-b7e8-b3dce873068d","year":2020},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.313536Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:dd9f20c72fbc5e62f6419cba733bd5b6035229e92f2e5ec58ffddbc5346e75a1","observation_id":"2400ff14-fab8-4a6c-8b88-69288592d0c8","resolution":{"observed_at":"2026-08-11T00:52:01.195587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.169348Z","title":"The interspeech 2020 deep noise suppres- sion challenge: Datasets, subjective testing framework, a nd challenge results,","venue":null,"work_id":"92a71da6-2288-430b-8955-78028482165b","year":2020},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.318285Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:7180ecad3041a495e664e2d887099fac8ed3a0c0c2c2033c143d01e2e66e1998","observation_id":"ab4a1127-1756-4997-9ed1-53ecbadbf73c","resolution":{"observed_at":"2026-08-11T00:52:01.176176Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.140264Z","title":"MP-SENet: A speech enhancement model with parallel denoising of magnitude and phase spectra,","venue":null,"work_id":"78cc4ed2-99f5-4678-b335-ff46cd878d97","year":2023},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.323500Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:b0458e68022d2fedc78c59185dd731efc0fb2d3e8fe2f014f25a76b9c94418ed","observation_id":"32dc520e-2811-44ed-bf88-3b2604c860b2","resolution":{"observed_at":"2026-08-11T00:52:01.147750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.119788Z","title":"Conv-TasNet: Surpassing idea l time–frequency magnitude masking for speech separation,","venue":null,"work_id":"437e1f17-381d-432b-b855-40a208626d40","year":2019},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.328670Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:2382a9bbbb625dc922c9cd802937d290a736b99629dd295534111633e60274e0","observation_id":"8cf34b5d-4ac9-4682-8ea6-2e91414d52b0","resolution":{"observed_at":"2026-08-11T00:52:01.126647Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.101226Z","title":"A time–frequency smoothing neural network for speech enhancement,","venue":null,"work_id":"2d1b38d5-fdf8-4254-b564-2dd1f42093d3","year":2020},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.333345Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:ee9008be41aef72a20ff8e38bb27e23e6c5eb0c51858c2b16c7601c425a8267b","observation_id":"ee3f4c51-2ae9-4b5a-8a82-568ae96ea6c5","resolution":{"observed_at":"2026-08-11T00:52:01.107689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.082495Z","title":"A perceptually-motivated approach for low-complexity, real-time enhancement of fullband speech ,","venue":null,"work_id":"bccad9f8-0472-45b4-a89e-c4720cde4864","year":2020},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.339947Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:d17050a948265708894310f5d8a8dcc5fce2ff5468a97e7186b09d97ea485b01","observation_id":"849031e7-5b10-40f7-9197-a285c2e1f07c","resolution":{"observed_at":"2026-08-11T00:52:01.088323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.08672","last_updated":"2021-06-16T10:16:35Z","snapshot_observed_at":"2026-08-19T12:19:47.516530Z","submitted_at":"2021-06-16T10:16:35Z","title":"DCCRN+: Channel-wise Subband DCCRN with SNR Estimation for Speech Enhancement","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.08672","snapshot_observed_at":"2026-08-11T00:52:00.346224Z","title":"DCCRN+: Channel-wise subband dccrn with snr estimation for speech enhancement,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.346224Z"},"links":{"cited_paper":"/paper/2106.08672","citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:cf6fd43dd20389a5c6dfae4523defb4d4eb005fcc445328c81cf0a80a15f229a","observation_id":"81089274-0dd3-4dd2-87fe-865d98715306","resolution":{"observed_at":"2026-08-11T00:52:00.346224Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.062704Z","title":"Lightweight full-band and sub-ba nd fusion network for real time speech enhancement,","venue":null,"work_id":"9af22e35-05c4-4d49-bc2b-5b41b206dbbb","year":2022},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.352434Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:479a56bbbdbe138f80e3c170ff8edc6a3bc240ee48d29e08b9155684b84b2c27","observation_id":"73119cd4-59b6-4db9-90f5-027640c16634","resolution":{"observed_at":"2026-08-11T00:52:01.069957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.040806Z","title":"OSSEM: One-shot speaker adaptive speech enhancement using meta learning,","venue":null,"work_id":"099f8638-1c33-40e9-9e7d-3174d3ce9bf4","year":2022},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.358502Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:f0a5cee3439e4ef0b29b51911399bf80bc1df3c62caedb44ff67414c7b36f65b","observation_id":"33d6cf69-1c7e-45c2-a585-f8542bb3cc44","resolution":{"observed_at":"2026-08-11T00:52:01.047431Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.018804Z","title":"WavLM: Large-scale self-supervised pre- training for full stack speech processing,","venue":null,"work_id":"91ebe27d-0448-457c-8fcf-8ab4c79612bc","year":2022},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.367094Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:50e436232d5f6da0e7e652e56fe94a6809853f489cb5653ad3d40b49e79866d5","observation_id":"ce81bb1f-6833-479f-9802-1896cb32f02a","resolution":{"observed_at":"2026-08-11T00:52:01.027231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.10388","last_updated":"2020-06-18T09:47:20Z","snapshot_observed_at":"2026-08-17T17:21:35.244412Z","submitted_at":"2020-06-18T09:47:20Z","title":"Self-supervised Learning for Speech Enhancement","version":1},"cited_work":{"arxiv_id":"2006.10388","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.10388","snapshot_observed_at":"2026-08-11T00:52:00.649503Z","title":"Self-supervised Learning for Speech Enhancement","venue":"eess.AS","work_id":"c29b0c11-33d2-473c-aa33-382bba1ca364","year":2020},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.372657Z"},"links":{"cited_paper":"/paper/2006.10388","citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:9675a0f0567cb92a49f875db552c8bc21aeecf6e55b3d0bb63427842a638c1e1","observation_id":"6d1b51aa-e99b-4998-b9a0-7aaaac729662","resolution":{"observed_at":"2026-08-11T00:52:00.655182Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.000498Z","title":"Self-supervised denoising autoencoder with linear regression decoder for speech enhancement,","venue":null,"work_id":"9666f59b-98d1-4e3b-a3d2-fdfd48aeb7b1","year":2020},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.379873Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:15a5e53edff97f9205cff95d5865ef80dc5b6b5bcf9a93937db89093f0bbf11a","observation_id":"6dbdb1f6-9df7-450a-bb24-1d9ed3ed0eda","resolution":{"observed_at":"2026-08-11T00:52:01.006440Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.978315Z","title":"Boosting self-supervised embeddings for speech enhancement,","venue":null,"work_id":"ee4ecd49-7df5-45a7-ac4a-ded97d19af5c","year":2022},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.386930Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:b59e9748956b6ec08c183db3dec7e8b2de8f996d4d40a050bcd5eaf547893af8","observation_id":"6b162383-a531-4cd1-b667-a97fc6190a3b","resolution":{"observed_at":"2026-08-11T00:52:00.986787Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.954606Z","title":"On generative spoken language modeling from raw audio,","venue":null,"work_id":"384fdfd8-f684-476b-b2b4-9cb9614fd9d7","year":2021},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.392886Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:fd74367c8bb90ee9779231f8b45ba2f2ad83d86a23fd973b5c5cd358e1f47268","observation_id":"79ffc082-b4fe-499a-9f25-681e7de718e3","resolution":{"observed_at":"2026-08-11T00:52:00.965992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.936496Z","title":"Analysing discrete self super vised speech representation for spoken language modeling,","venue":null,"work_id":"6ea33958-5090-43a7-8d8d-8cdeb501d441","year":2023},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.399945Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:9cae27a61c57befd2bdd6f678d1013e0616fd50c99fe48054080465ff25a8a74","observation_id":"2d8ad8df-8085-4e76-8839-48ce7140bcf3","resolution":{"observed_at":"2026-08-11T00:52:00.942187Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.918203Z","title":"SELM: Speech enhancement using discrete tokens and language models,","venue":null,"work_id":"9f695d05-9ed9-4795-84fb-65f7a036d7af","year":2024},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.404767Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:b234bf975e7c5e72fcf948dc94de4fe046c38d6094e17c69c3e826cf53b9a91d","observation_id":"a4f53eb0-3ea5-4465-9b88-80046776beea","resolution":{"observed_at":"2026-08-11T00:52:00.924195Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.00704","last_updated":"2024-12-10T03:17:56Z","snapshot_observed_at":"2026-08-16T14:56:06.586855Z","submitted_at":"2023-10-01T15:49:46Z","title":"UniAudio: An Audio Foundation Model Toward Universal Audio Generation","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.00704","snapshot_observed_at":"2026-08-11T00:52:00.409922Z","title":"Uniaudio: An audio foundation model toward universal audio generation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.409922Z"},"links":{"cited_paper":"/paper/2310.00704","citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:7e7071cdb99232437a408484e8d83971a5e77e776fa26d54bdbb8c73a83aa287","observation_id":"de2112e8-330e-4845-ba48-d41723859f7e","resolution":{"observed_at":"2026-08-11T00:52:00.409922Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.900057Z","title":"Speechx: Neural codec language model as a versatile speech transformer,","venue":null,"work_id":"e0c82828-1c4c-408b-8aec-8d18c49046d5","year":2024},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.416269Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:935c3e1a1c5fb19852d0781ca0db2c2bf478f894e3cf862de946e9d3915cc333","observation_id":"eacd1cd2-5973-4566-9e60-6894f9e1bb43","resolution":{"observed_at":"2026-08-11T00:52:00.905524Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.884180Z","title":"Low-latenc y incremental text-to-speech synthesis with distilled cont ext prediction network,","venue":null,"work_id":"ac980f66-13f1-4860-aded-ca522d805156","year":2021},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.421589Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:d13b6ac5ee5caf3a6428ec1ece97037621c3fb50cb0d6e38426366bf08f9f516","observation_id":"3227b3cd-2699-4297-9068-62b2688076cb","resolution":{"observed_at":"2026-08-11T00:52:00.889762Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.865127Z","title":"DualVC 3: Leveraging language model gen- erated pseudo context for end-to-end low latency streaming voice conversion,","venue":null,"work_id":"b95e073e-73b9-46e6-b01e-d75284302ac2","year":2024},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.427458Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:2e35dc8d4e25c291b0283725a964fb25f9e6ef7984d6d579d18e92443ce86f16","observation_id":"04bf711c-b779-4ac8-bc2c-9ad18141eaeb","resolution":{"observed_at":"2026-08-11T00:52:00.871896Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.841290Z","title":"The voice bank corp us: Design, collection and data analysis of a large regional acc ent speech database,","venue":null,"work_id":"d442c1be-1f91-4c3f-934c-72d2e36bfdce","year":2013},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.434170Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:4bf1a35e2895ec650a91d98e748b33cb5ca2080f6e2dce71b5e84839c7a93147","observation_id":"af42f6a9-4d61-4bbb-9759-bc780c4708dc","resolution":{"observed_at":"2026-08-11T00:52:00.847719Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.823224Z","title":"The diverse enviro n- ments multi-channel acoustic noise database (demand): A database of multichannel environmental noise recordings,","venue":null,"work_id":"dd612192-86c0-44d3-868e-300c53a47210","year":2013},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.440785Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:220fd2735eaacca478225a44d3ae25142f0a3c8dbaab5d164aa5350145f5109a","observation_id":"b30ddaa7-0d7b-4d40-b386-efc6fbd3f3b6","resolution":{"observed_at":"2026-08-11T00:52:00.828753Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.806354Z","title":"Investigating rnn-based speech enhancement methods for noise-robust text-to-speech.,","venue":null,"work_id":"6306bb6e-2f99-4f89-8b8b-b2ad343ca90e","year":2016},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.447066Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:d8915fd274d17c94f5c99ca41320ae1cfa46f0588075035f639e2f57575f2651","observation_id":"91e44e7b-06bc-4daa-abc3-be6d853ab5e7","resolution":{"observed_at":"2026-08-11T00:52:00.811640Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.789030Z","title":"Boosting objective scores of a speech enhancement model by metricgan post-processing,","venue":null,"work_id":"fb3e4dcc-bb29-47aa-910a-9e3debe66829","year":2020},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.453132Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:b95b3936c25656ea15c9adfe975aae53100f7f9e74853f19d9e9d22d11c0da47","observation_id":"f8d080e0-7e66-4a48-9b1f-65d93bcbb844","resolution":{"observed_at":"2026-08-11T00:52:00.794891Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.769795Z","title":"Run-and-back stitch search: Novel block synchronous decoding for streaming encoder-decoder ASR,","venue":null,"work_id":"c94e1ead-356a-4cf7-928a-a12448c2b4c5","year":2022},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.459633Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:933bfafb2e0d4f54e15bdc45ded67e9ae038aff3afb993cd095c8e284fb7e7d0","observation_id":"b413a75a-af94-41b8-a8e0-6d45bb11ee3a","resolution":{"observed_at":"2026-08-11T00:52:00.776280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.750535Z","title":"Attention is all you need,","venue":null,"work_id":"919a22bd-bf3f-4dd1-8d33-f2fbfbe34365","year":2017},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.464397Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:94749c10d2a15d2a63d526d316fe87653a28d99c5155b3c32c291779b4c3c93e","observation_id":"08edef8c-21fd-4b8d-8b88-70942d38ea4b","resolution":{"observed_at":"2026-08-11T00:52:00.755969Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.729178Z","title":"FiLM: Visual reasoning with a general conditioning layer,","venue":null,"work_id":"1404a25c-bb56-44d3-90ee-fcbfe9e2d6b0","year":2018},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.471249Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:db52284e579d976808e7a2563d5734ac061213ddd1c88d17cca76a248d774775","observation_id":"b83fc670-6bc5-4283-96d4-a545d86e7ccf","resolution":{"observed_at":"2026-08-11T00:52:00.737350Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.00169","last_updated":"2024-07-22T09:53:44Z","snapshot_observed_at":"2026-08-17T17:21:34.645323Z","submitted_at":"2023-08-31T23:26:10Z","title":"RepCodec: A Speech Representation Codec for Speech Tokenization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.00169","snapshot_observed_at":"2026-08-11T00:52:00.476175Z","title":"RepCodec: A speech representation codec for speech tokenization,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.476175Z"},"links":{"cited_paper":"/paper/2309.00169","citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:3b76955c8b4ddbb800a1364b13cb23263b8e87ed0e9c0cdb6806e5ac7204ac71","observation_id":"c37cd139-7a05-4392-b4b7-78c605da9ebf","resolution":{"observed_at":"2026-08-11T00:52:00.476175Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.706547Z","title":"Generating diverse high-ﬁdelity images with VQ-V AE-2,","venue":null,"work_id":"efa38b7b-0a2c-4e0b-9ef6-65bee055e362","year":2019},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.481481Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:64690b06f3b3f2ccd4bfd1b6a6f0bb04e4869749e005d26525883c166f47eb43","observation_id":"0c57409c-2123-4000-8ba9-8a04f5ba34aa","resolution":{"observed_at":"2026-08-11T00:52:00.713196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.08551","last_updated":"2025-05-27T05:07:56Z","snapshot_observed_at":"2026-08-17T17:09:21.212561Z","submitted_at":"2024-07-11T14:36:53Z","title":"Autoregressive Speech Synthesis without Vector Quantization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.08551","snapshot_observed_at":"2026-08-11T00:52:00.486868Z","title":"Autoregressive speech synthesis without vector quantization,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.486868Z"},"links":{"cited_paper":"/paper/2407.08551","citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:1b9289f355e323b1164d9a50215b4a98681cf2642d4bde31036e6761fd9f7000","observation_id":"ae2ee173-6146-4b39-a091-42b0cedb1bdb","resolution":{"observed_at":"2026-08-11T00:52:00.486868Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:00.685628Z","title":"Evaluation of objective quality measures for speech enhancement,","venue":null,"work_id":"8a4a30c7-82c4-4e81-8ae7-ebaf36a00698","year":2007},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.493074Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:e28b3abc0924bc9db4196895a962cbfebd7724dae038ac2d9b4888c35994dc0f","observation_id":"1ff297ac-ceaa-48ca-8f2a-584a9ea323c9","resolution":{"observed_at":"2026-08-11T00:52:00.693386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.11611","last_updated":"2020-08-20T05:12:35Z","snapshot_observed_at":"2026-08-17T17:21:18.765041Z","submitted_at":"2020-05-23T22:17:49Z","title":"Exploring the Best Loss Function for DNN-Based Low-latency Speech Enhancement with Temporal Convolutional Networks","version":3},"cited_work":{"arxiv_id":"2005.11611","doi":null,"metadata_source":"pith","pith_arxiv_id":"2005.11611","snapshot_observed_at":"2026-08-11T00:52:00.555346Z","title":"Exploring the Best Loss Function for DNN-Based Low-latency Speech Enhancement with Temporal Convolutional Networks","venue":"eess.AS","work_id":"1bdcfa36-54c4-4bde-9249-a39a47748aa3","year":2020},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.500387Z"},"links":{"cited_paper":"/paper/2005.11611","citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:d8fba4c7afeccd0cd8a298d4be424b3ed5abfb8a4501af2c72c72c5280fc8c41","observation_id":"54e18c4c-abb7-4583-b8dc-9904a50ea83e","resolution":{"observed_at":"2026-08-11T00:52:00.565646Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:52:01.247528Z","title":"A convolutional recurrent neural network for real-time speech enhancement,","venue":null,"work_id":"934bcf2d-bcc4-492f-bf0d-86fef397082f","year":2018},"citing_paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-11T00:52:00.506822Z"},"links":{"citing_paper":"/paper/2412.19248"},"observation_digest":"sha256:ca50b36392954330b0fe19b5275d11909ac536dacbdbb73fdbfa2c9428a628c8","observation_id":"db4c6660-4363-48ae-882c-3910610a5500","resolution":{"observed_at":"2026-08-11T00:52:01.253451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2412.19248","last_updated":"2024-12-26T15:08:36Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-17T17:11:14.624276Z","submitted_at":"2024-12-26T15:08:36Z","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":4,"verified_exact":2,"verified_fuzzy":32},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 22 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 0 inbound Pith citation observations for arXiv:2412.19248."}