{"as_of":"2026-08-08T07:21:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b2c48272cde9b6785a5645ea89bfee437fcfcbd8b3a46c1948303f17ab27cf26","coverage":[{"denominator":54,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":54,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T00:59:56.303718Z","state":"measured"},{"denominator":55,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":55,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T21:34:07.924156Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-06T21:34:11.220887Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"cited_work":{"arxiv_id":"2506.12260","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.12260","snapshot_observed_at":"2026-08-06T21:34:11.220887Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","venue":"cs.SD","work_id":"066ac648-864a-4e98-ab50-bc1232e0ab66","year":2025},"citing_paper":{"arxiv_id":"2506.23859","last_updated":"2025-08-19T09:17:36Z","snapshot_observed_at":"2026-08-06T21:27:35.449529Z","submitted_at":"2025-06-30T13:55:10Z","title":"Less is More: Data Curation Matters in Scaling Speech Enhancement","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T21:34:07.924156Z"},"links":{"cited_paper":"/paper/2506.12260","citing_paper":"/paper/2506.23859"},"observation_digest":"sha256:90694324107c9f2fa1b0ac8c0841a168e4cf69beb2d42c85406413699ea94584","observation_id":"3fbac6c8-1f69-4e11-b181-9c5c61e0a827","resolution":{"observed_at":"2026-08-06T21:34:11.310798Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.12260/citation-record","integrity":"/paper/2506.12260/integrity","json":"/paper/2506.12260/citation-record.json","paper":"/paper/2506.12260"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:51.318929Z","title":null,"venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:51.318929Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:781b305e08c104fa55bd8cf41d342e403528963170b900411576e75db9bf7d9b","observation_id":"b28c77b7-c622-455f-8923-0c0497d98042","resolution":{"observed_at":"2026-08-07T00:59:51.318929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:02.784094Z","title":"SDR–half- baked or well done?","venue":null,"work_id":"9f7557ad-4788-4b1f-9edb-119c7d8afdfe","year":2019},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:51.393421Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:5f2b0466e7af94701a8414aaf9346240630b83bd9b3632cd12905da4188e1ebc","observation_id":"cf9aca71-7c68-471f-8309-c9ab7386e0d9","resolution":{"observed_at":"2026-08-07T01:00:02.810913Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:02.704834Z","title":"How bad are artifacts?: Analyzing the impact of speech enhancement errors on asr,","venue":null,"work_id":"b6c36e56-46ad-4996-9651-e9ec5090d19a","year":2022},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:51.494513Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:f356ba2e6d22ba2aefb1a58a179b3239309c4c529f51d718c44150eafc98afba","observation_id":"21a0fe25-e3cb-480e-8759-75a7212852d2","resolution":{"observed_at":"2026-08-07T01:00:02.732807Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:02.640903Z","title":"Bridging the gap between monaural speech enhancement and recognition with distortion-independent acous- tic modeling,","venue":null,"work_id":"3398b4f9-a5df-4386-8960-afdcd2b74e42","year":2020},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:51.592392Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:b4623508e63f39f159cee2ae177c3d6d07acee47eb5327588d65538b35cf0f7d","observation_id":"462bd6d8-9579-48ab-8a2b-b97a2c17bb4b","resolution":{"observed_at":"2026-08-07T01:00:02.663269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:02.583223Z","title":"Advancing non-intrusive suppression on enhancement distortion for noise robust asr,","venue":null,"work_id":"dcf59929-2c15-4006-b0db-39ab04d20bbb","year":2025},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:51.731425Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:e698f1591fc485ac8e4797117c9583cbb94c231f48fe921bf7611a36cedb1e67","observation_id":"18d0c539-84f8-47f7-a0f3-de04eb4b5ebe","resolution":{"observed_at":"2026-08-07T01:00:02.606075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:02.529131Z","title":"Fat-hubert: Front-end adaptive training of hidden-unit bert for distortion-invariant robust speech recognition,","venue":null,"work_id":"3a771e6d-b5d6-4e0e-bb71-6c5500820466","year":2023},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:51.863593Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:ae8c2bad9914c9244ea4572807e57529960a98ce7885523c426c7a4fd681e70a","observation_id":"2c90d128-7dfd-4885-bdaa-c7e8af5ba188","resolution":{"observed_at":"2026-08-07T01:00:02.555239Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:02.467084Z","title":"Closing the gap between time-domain multi-channel speech enhancement on real and simulation conditions,","venue":null,"work_id":"b247b830-0711-4d64-8760-d15643d95cdd","year":2021},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:52.007591Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:08b5927b83d9bc95068455887fa987af712e3052766eec203a9b76a81cf2e3c8","observation_id":"aa56e567-42db-46b9-8ad8-67614693cccb","resolution":{"observed_at":"2026-08-07T01:00:02.496086Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23859","last_updated":"2025-08-19T09:17:36Z","snapshot_observed_at":"2026-08-06T21:27:35.449529Z","submitted_at":"2025-06-30T13:55:10Z","title":"Less is More: Data Curation Matters in Scaling Speech Enhancement","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.23859","snapshot_observed_at":"2026-08-07T00:59:52.138592Z","title":"Less is more: Data curation matters in scaling speech enhancement,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:52.138592Z"},"links":{"cited_paper":"/paper/2506.23859","citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:402979684827a1139ed83f8932cc5c8514f60a01ef3b3178a6553f7abd2452c3","observation_id":"799acbfa-1e7f-4781-8821-d5544ec02d31","resolution":{"observed_at":"2026-08-07T00:59:52.138592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:02.307482Z","title":"Lightweight Front-end Enhancement for Robust ASR via Frame Resampling and Sub-Band Pruning,","venue":null,"work_id":"325baf87-6147-436d-94f0-e98b12a63166","year":2025},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:52.248087Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:3a2d5d45d2ddff3b121132f0b1979b57aaf8b9477c5886a65bb438cc7a04a994","observation_id":"96a43799-7db9-4a3f-874e-205fe6af9ebb","resolution":{"observed_at":"2026-08-07T01:00:02.437242Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:02.044149Z","title":"A review on subjective and objective evaluation of synthetic speech,","venue":null,"work_id":"de423ff6-9a2e-4ce0-ad35-3a026f12c318","year":2024},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:52.384331Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:2a60693ee95d7cd339bfd5dc73f09b6dd76eaae0e5555cfc8a3765b40c607040","observation_id":"71ce7602-6300-4836-a3d3-a682f31ac579","resolution":{"observed_at":"2026-08-07T01:00:02.156596Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:01.855739Z","title":"Objective measures of perceptual audio quality reviewed: An evaluation of their application domain dependence,","venue":null,"work_id":"17f0a796-ef09-4539-b24e-163d09a606d9","year":2021},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:52.526177Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:4f970bd95b4c5efb860000c7c3530d4857d22c9f26ee8067ba17fa5b411e5657","observation_id":"9b077439-c7e1-4fdf-9589-6e8474f368cc","resolution":{"observed_at":"2026-08-07T01:00:01.982525Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:01.638559Z","title":"Versa: A versatile evaluation toolkit for speech, audio, and music,","venue":null,"work_id":"98636bb1-9154-4349-ad48-33e3ac52555e","year":2024},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:52.638340Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:5b34f1ee48a1e1c78cd0a36d88e07ebf9631610b6a8787d084d595ac938ea666","observation_id":"3e5aa4c9-92f2-4de0-b13b-3d3faf4c71ad","resolution":{"observed_at":"2026-08-07T01:00:01.716700Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01611","last_updated":"2025-06-02T12:50:37Z","snapshot_observed_at":"2026-08-07T11:35:03.897104Z","submitted_at":"2025-06-02T12:50:37Z","title":"Lessons Learned from the URGENT 2024 Speech Enhancement Challenge","version":1},"cited_work":{"arxiv_id":"2506.01611","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.01611","snapshot_observed_at":"2026-08-07T00:59:56.760465Z","title":"Lessons Learned from the URGENT 2024 Speech Enhancement Challenge","venue":"eess.AS","work_id":"af144f5f-e926-40a2-b4f7-00508ad4f919","year":2025},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:52.754052Z"},"links":{"cited_paper":"/paper/2506.01611","citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:1ac4284e0990d6ab4034e5c063b15ea5f4c1529f8a35890f0060f60b0f3fc01c","observation_id":"5d88ab18-9866-414a-93e4-188c6102559e","resolution":{"observed_at":"2026-08-07T00:59:56.812979Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.05139","last_updated":"2025-02-07T18:15:57Z","snapshot_observed_at":"2026-07-06T20:33:01.960703Z","submitted_at":"2025-02-07T18:15:57Z","title":"Meta Audiobox Aesthetics: Unified Automatic Quality Assessment for Speech, Music, and Sound","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.05139","snapshot_observed_at":"2026-08-07T00:59:52.863711Z","title":"Meta audiobox aesthetics: Unified automatic quality assessment for speech, music, and sound,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:52.863711Z"},"links":{"cited_paper":"/paper/2502.05139","citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:2ad9bee5e84f38e4d2fd0f6b50f1a9e12d9fa7872c8beff5e683d8e1b672eaca","observation_id":"9fda4650-5519-426c-ad6a-23f20fcf6493","resolution":{"observed_at":"2026-08-07T00:59:52.863711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:01.429502Z","title":"DNSMOS P.835: A non-intrusive perceptual objective speech quality metric to evaluate noise suppressors,","venue":null,"work_id":"3b547d78-8c3a-43d9-a20c-147c668147fa","year":2022},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:52.978436Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:6ec8b22107d1460ebd1319d146616652d05c6e65d394cf76abe5b60bcec32fac","observation_id":"80f9f217-b210-4834-896a-ec540a5dcbd5","resolution":{"observed_at":"2026-08-07T01:00:01.527396Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:01.174139Z","title":"UTMOS: UTokyo-SaruLab system for V oiceMOS chal- lenge 2022,","venue":null,"work_id":"cdd69636-ff91-4670-9693-449637bb36c6","year":2022},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:53.142609Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:297faea1615acf1dc2fb271eb8c6d0398ca21cd467eb9a56cfa52138cac790a2","observation_id":"27c49265-0fba-42b5-bd78-2e6f52b627d0","resolution":{"observed_at":"2026-08-07T01:00:01.324060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:00.804801Z","title":"The t05 system for the voicemos challenge 2024: Transfer learning from deep image classifier to naturalness mos prediction of high-quality synthetic speech,","venue":null,"work_id":"ffe6b44a-93ea-49b8-a93a-1516272dcabb","year":2024},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:53.243207Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:f1ae19c2067fb60c3de8cfa316a9e689e70e614b60d99095a7fa50733ad97715","observation_id":"97c161ad-e9fe-4bc3-9e0e-38510567681c","resolution":{"observed_at":"2026-08-07T01:00:01.010390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:00.497999Z","title":"The voicemos challenge 2022,","venue":null,"work_id":"09deaa18-07c6-4bbf-958e-bf640fd0fc64","year":2022},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:53.351630Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:297fc1375a58a896e34c23eceeb130ffaac92cffe91c158c783eaa691076de3a","observation_id":"526ae2a3-fbfd-4f51-bcc8-6f7dc5585c70","resolution":{"observed_at":"2026-08-07T01:00:00.617091Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:00.330324Z","title":"The voicemos challenge 2023: Zero-shot subjective speech quality prediction for multiple domains,","venue":null,"work_id":"fef8a573-0a63-4e72-a682-660955e926da","year":2023},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:53.468606Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:1284912eb76ea1711c4571c11a1e86539ba409a0a5fc77d640417183acba4244","observation_id":"73aad329-0b7a-4d60-9102-eb326a1451c0","resolution":{"observed_at":"2026-08-07T01:00:00.399408Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:00.174318Z","title":"The voicemos challenge 2024: Beyond speech quality prediction,","venue":null,"work_id":"2b12063a-8cfa-4c9c-8925-7420d3388de3","year":2024},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:53.604460Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:b063b686f5b8c30613c1650eb83ab3d8326523e91e9642cdf2483c21b9a874c7","observation_id":"b1911d94-4095-45af-a45c-c3fd70cc2894","resolution":{"observed_at":"2026-08-07T01:00:00.262689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23874","last_updated":"2025-06-30T14:05:17Z","snapshot_observed_at":"2026-08-06T21:27:11.504822Z","submitted_at":"2025-06-30T14:05:17Z","title":"URGENT-PK: Perceptually-Aligned Ranking Model Designed for Speech Enhancement Competition","version":1},"cited_work":{"arxiv_id":"2506.23874","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.23874","snapshot_observed_at":"2026-08-07T00:59:56.633705Z","title":"URGENT-PK: Perceptually-Aligned Ranking Model Designed for Speech Enhancement Competition","venue":"eess.AS","work_id":"95cfa651-bc11-405c-852c-9e1c71f5a631","year":2025},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:53.700804Z"},"links":{"cited_paper":"/paper/2506.23874","citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:932718fdaadc647d53b48f61f72dbcbe2f884677258b059b5b1631496d54b500","observation_id":"828a28ec-a6a1-4453-be17-03e14ef6c3ef","resolution":{"observed_at":"2026-08-07T00:59:56.682451Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:00:00.050869Z","title":"ICASSP 2024 speech signal improvement challenge,","venue":null,"work_id":"a6d1335e-8fc6-48bc-82af-a212d322883c","year":2024},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:53.768549Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:4951e70ddab834760cecc38ac3dcfa1d44fee934ae65f666c994cf00814e48b5","observation_id":"1610e23b-d9f4-4649-8891-e3c20ab1fbde","resolution":{"observed_at":"2026-08-07T01:00:00.101072Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.20741","last_updated":"2025-05-27T05:31:19Z","snapshot_observed_at":"2026-08-07T23:52:33.743427Z","submitted_at":"2025-05-27T05:31:19Z","title":"Uni-VERSA: Versatile Speech Assessment with a Unified Network","version":1},"cited_work":{"arxiv_id":"2505.20741","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.20741","snapshot_observed_at":"2026-08-07T00:59:56.523759Z","title":"Uni-VERSA: Versatile Speech Assessment with a Unified Network","venue":"cs.SD","work_id":"cdb00c1b-7340-444c-b69a-2e1b50468391","year":2025},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:53.899052Z"},"links":{"cited_paper":"/paper/2505.20741","citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:b6b6fb67e970cf32db6e1864d720d7b364a3fdd51d7e881af4403164c9e22191","observation_id":"64a7a598-988f-45c9-8b04-891bd755a587","resolution":{"observed_at":"2026-08-07T00:59:56.557082Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:59.874908Z","title":"Perceptual evaluation of speech quality (PESQ)—a new method for speech quality assessment of telephone networks and codecs,","venue":null,"work_id":"b3987799-dfb0-4cf5-95bb-198afe09c905","year":2001},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:54.030231Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:99e5086c4e40c9d01b61521a5d8bdfeb73403f30d87df4db4b69e0e90e3b48da","observation_id":"c6de50a5-1516-4526-a048-fd359071ccdd","resolution":{"observed_at":"2026-08-07T00:59:59.954336Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:59.690046Z","title":"Perceptual objective listening quality assess- ment (POLQA), the third generation ITU-T standard for end-to-end speech quality measurement part I–—temporal alignment,","venue":null,"work_id":"5330cf9a-34cb-4af8-bade-9bef379c524c","year":2013},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:54.165739Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:0f7dc7854bff26b02360e48a653e54c2d7bf45de3cbfdc8500b8a29d2c274537","observation_id":"717af5b2-ce14-497d-a98e-e3e55a746213","resolution":{"observed_at":"2026-08-07T00:59:59.760642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:59.547647Z","title":"URGENT challenge: Universality, robustness, and generalizability for speech en- hancement,","venue":null,"work_id":"9ccdd92f-2989-4421-916a-89274503ee6d","year":2024},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:54.277950Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:26964d37b79b59f0920ca7c55971ac51a4d0be662550277a0679b22d5f8ab9f9","observation_id":"286b5014-978a-4f2d-a3a5-d389b67e1ec7","resolution":{"observed_at":"2026-08-07T00:59:59.624900Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:59.422542Z","title":"Inter- speech 2025 URGENT speech enhancement challenge,","venue":null,"work_id":"5146e0bd-e107-4c78-b74c-af42e651b18b","year":2025},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:54.413312Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:a04706deb07125bc64a7c874328210838ee4c82a86f6e95b5f3fd472702cd660","observation_id":"b3bbf6c2-0e84-4dc6-8791-bbab95095f28","resolution":{"observed_at":"2026-08-07T00:59:59.472832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:59.295446Z","title":"Performance measurement in blind audio source separation,","venue":null,"work_id":"ec524388-313e-4252-9f57-c8c9092336c2","year":2006},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:54.511671Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:7220772cb696e0185382981bb6fc358dd54ebaafa4e0e40e12770ac30b130568","observation_id":"13f11994-095c-42dd-ac5b-d32284d9a7c2","resolution":{"observed_at":"2026-08-07T00:59:59.343918Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:59.124792Z","title":"Distillation and pruning for scalable self- supervised representation-based speech quality assessment,","venue":null,"work_id":"5f7cdd23-6c0a-486f-8143-5a14f83d273f","year":2025},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:54.648344Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:3c240f90c651085a1bfda149ef571c351a8ec86c64684dc4961f808489d08007","observation_id":"78437f9c-7a8c-442d-8745-7c857bbeb1ab","resolution":{"observed_at":"2026-08-07T00:59:59.213900Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:58.975232Z","title":"NISQA: A deep CNN- self-attention model for multidimensional speech quality prediction with crowdsourced datasets,","venue":null,"work_id":"876db6ed-0e11-4aef-bc7a-3615bb9d661d","year":2021},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:54.745525Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:e2f6c5850d95f5aa18c36a785b909e144de3098ff9e1132a2122dac16f148134","observation_id":"daf7cdf0-5a0b-4cd1-a709-c00337bfc8d7","resolution":{"observed_at":"2026-08-07T00:59:59.041433Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:58.785907Z","title":"SCOREQ: Speech quality assessment with contrastive regression,","venue":null,"work_id":"1c8651ea-37b3-4e22-88ea-ee89d37c2bd6","year":2024},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:54.832101Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:7c812d6c9a94f9d0d11a85a5483360750ec0fdb15a578a1a87f9ea47761a298e","observation_id":"963dbb25-bd7e-4808-9049-20944ecea082","resolution":{"observed_at":"2026-08-07T00:59:58.879269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:58.643919Z","title":"Owsm v3. 1: Better and faster open whisper-style speech models based on e-branchformer,","venue":null,"work_id":"978089c7-27c4-4ed9-ae5b-b3987dba6e03","year":2024},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:54.923054Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:9f337f94f404e45ad774dbe83f22de25c3b4e7b4618915203c85f09b61cd4234","observation_id":"0faae484-a6a3-4fa9-b9cf-a0852255993b","resolution":{"observed_at":"2026-08-07T00:59:58.708717Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:58.541765Z","title":"An algorithm for predicting the intelligibility of speech masked by modulated noise maskers,","venue":null,"work_id":"23b29705-df92-4b92-bbdb-ae5aa90310a2","year":2009},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:55.025151Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:08f421f44b28ce3faff7bc53d6e2806fabcca3f0b45b708f6aa954b31f7caf61","observation_id":"39fe393c-1152-4e33-bc00-37f251892288","resolution":{"observed_at":"2026-08-07T00:59:58.586149Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:58.458714Z","title":"SpeechBERTScore: Reference-aware automatic evaluation of speech generation leveraging NLP evaluation metrics,","venue":null,"work_id":"de2e3d8e-0d68-4609-92d6-f0d734ba5e1d","year":2024},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:55.155079Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:522fc59c3568334e1279ae074c02c237408703d8440301884fee0a24bbb9bfbc","observation_id":"7be65e64-7583-49de-98a1-d3ab0c68ef69","resolution":{"observed_at":"2026-08-07T00:59:58.488141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:55.245748Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:55.245748Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:65a270ca894bb7faf1544bbbe5effb77782462f32827b27271f72addefc54645","observation_id":"5acf0661-06e7-448f-80d5-79851f3403fe","resolution":{"observed_at":"2026-08-07T00:59:55.245748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:58.299665Z","title":"Evaluation metrics for generative speech enhancement methods: Issues and perspectives,","venue":null,"work_id":"8981e35c-4a5b-4938-9b1f-576c9a764840","year":2023},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:55.351874Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:97260c745ca1b8f7db16189ab3a4eb3a607759a2801c563fab4d9a384c3fdfd9","observation_id":"4abb97a3-c6e6-43af-b5f2-a7c70b786a03","resolution":{"observed_at":"2026-08-07T00:59:58.367515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:58.132328Z","title":"Espnet-spk: full pipeline speaker embedding toolkit with reproducible recipes, self- supervised front-ends, and off-the-shelf models,","venue":null,"work_id":"55998d1e-1192-4674-848b-a475f0f23bf9","year":2024},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:55.433303Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:c92920ee4d22cbf88908e22ea29bcfafcf31f65a695a4188d24989efeaafe421","observation_id":"ba36eb9c-7f0d-4ea1-bdc6-2d178ed17c46","resolution":{"observed_at":"2026-08-07T00:59:58.188012Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:58.004147Z","title":"Mel-cepstral distance measure for objective speech qual- ity assessment,","venue":null,"work_id":"2d4f79b3-61c4-47ba-ad33-041b8f1e630b","year":1993},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:55.488912Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:9b9b334d7d6e6b2fdd6f6b3c68742d96313134d02131056924c7de26f3fd153d","observation_id":"825bc888-e9c2-43d5-bc37-41bf4fb89d03","resolution":{"observed_at":"2026-08-07T00:59:58.069927Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:57.838552Z","title":"Lessons learned from the URGENT 2024 speech enhancement chal- lenge,","venue":null,"work_id":"3c2ad75a-da88-4281-ae34-bdf61ca08349","year":2024},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:55.528778Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:9bffc40a0be79414026dcd2473a593951b065ed0ffddb1267b163b15f1f61946","observation_id":"f53e7362-7fc5-4143-8137-8d17d957a590","resolution":{"observed_at":"2026-08-07T00:59:57.928233Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:57.729300Z","title":"Distance measures for speech processing,","venue":null,"work_id":"5bcd6ee4-fbed-4184-b9dc-2a8d3392f005","year":1976},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:55.577652Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:f14067f3eeb5b124d36850ee3b8b2b34940f2d22ec0b8b202a5335eeaf5c6086","observation_id":"f8d7e008-305d-4c47-a548-798d160bd59d","resolution":{"observed_at":"2026-08-07T00:59:57.771171Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15061","last_updated":"2025-05-21T03:30:23Z","snapshot_observed_at":"2026-08-07T15:22:42.403133Z","submitted_at":"2025-05-21T03:30:23Z","title":"SHEET: A Multi-purpose Open-source Speech Human Evaluation Estimation Toolkit","version":1},"cited_work":{"arxiv_id":"2505.15061","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15061","snapshot_observed_at":"2026-08-07T00:59:56.411210Z","title":"SHEET: A Multi-purpose Open-source Speech Human Evaluation Estimation Toolkit","venue":"cs.SD","work_id":"7c98b9b2-c521-450b-8def-4d2fc50711bd","year":2025},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:55.643986Z"},"links":{"cited_paper":"/paper/2505.15061","citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:e7c8fbb3dedd4755bd704e98f5901fb4858d1e9bc891084b3d78411868b8cfb8","observation_id":"7b40b310-2311-414f-955d-0e1c5e80c18d","resolution":{"observed_at":"2026-08-07T00:59:56.456405Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:57.575877Z","title":"The chime-7 udase task: Unsupervised domain adaptation for conversational speech enhancement,","venue":null,"work_id":"a38b05a9-1ba1-4551-9541-40834ab4f18e","year":2023},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:55.698323Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:057bad0b289d4c70e531c3b7a3e169a2cf97e18afe6647c2ca27981998c7a95a","observation_id":"3013255a-d3d3-4827-bc4f-2c9f25c1cac1","resolution":{"observed_at":"2026-08-07T00:59:57.667117Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:57.465390Z","title":"Generalization ability of mos prediction networks,","venue":null,"work_id":"ceb1e584-5ce5-4a81-ad66-237c986c7d93","year":2022},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:55.749707Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:560a19e770b88de0137fe8851f0dd838b96ef47c9da534622da230d02afc783c","observation_id":"fba4b467-5ee4-4b4f-8a1b-3e52edcded0e","resolution":{"observed_at":"2026-08-07T00:59:57.516889Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:57.331298Z","title":"The blizzard challenge 2019,","venue":null,"work_id":"e5dec4ca-b4cb-4c74-94c1-76a778f3dcfe","year":2019},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:55.796894Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:bfbd20a8d39a1463fc50f07755219b509dcc15ef33e848ec21e4f96be4d5be0b","observation_id":"dcf98de6-a306-46dd-a800-20bd7f478278","resolution":{"observed_at":"2026-08-07T00:59:57.400223Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:55.841645Z","title":"Mos-bench: Benchmarking generalization abilities of subjective speech quality assessment models,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:55.841645Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:615992257a197c90cae7cf4e70140bb15c51c56e57f54dbe185c1f4d735c31b9","observation_id":"6cc6fd5e-383a-4eef-a2d3-8f238e7618a1","resolution":{"observed_at":"2026-08-07T00:59:55.841645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:57.212467Z","title":"The INTERSPEECH 2020 deep noise suppression challenge: Datasets, subjective testing framework, and challenge results,","venue":null,"work_id":"1ec21386-60ea-413a-82a3-be45f60fa5eb","year":2020},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:55.967591Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:78063edd3bdbd0c7487ab998c461ababe46c9c9c5552c7f51ac707219c6db4c4","observation_id":"57d83909-e1ff-4e8e-9e5e-0ce9d9f0bbdd","resolution":{"observed_at":"2026-08-07T00:59:57.250392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:57.124243Z","title":"An analysis of environment, microphone and data simulation mismatches in robust speech recognition,","venue":null,"work_id":"25b64b03-9ba5-4d58-a57b-a43b43cb50c6","year":2017},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:56.011864Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:4087180f33a0bb6e53e7d3b65cd5f930edf2b26abd43a1100a0e2dcf1f574d2e","observation_id":"26c706da-74de-42f9-9042-ce318aa2c9fb","resolution":{"observed_at":"2026-08-07T00:59:57.155828Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1804.00015","last_updated":"2018-03-30T18:09:39Z","snapshot_observed_at":"2026-07-06T06:31:07.605442Z","submitted_at":"2018-03-30T18:09:39Z","title":"ESPnet: End-to-End Speech Processing Toolkit","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1804.00015","snapshot_observed_at":"2026-08-07T00:59:56.058833Z","title":"Espnet: End-to-end speech processing toolkit,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:56.058833Z"},"links":{"cited_paper":"/paper/1804.00015","citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:8114be1a82e33babe5141f04179d94a9557ebee21c64b13a080cf4a37bb3c719","observation_id":"7dd16f2b-a7b8-4757-87ff-b1e5e21a8ad7","resolution":{"observed_at":"2026-08-07T00:59:56.058833Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:56.102596Z","title":"Wavlm: Large-scale self-supervised pre- training for full stack speech processing,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:56.102596Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:a2ace923065b46f825f2218fc1f60323835ae351b5d328fec659595ccd2162f8","observation_id":"b99c0718-0b12-4817-8cb4-94b26172514d","resolution":{"observed_at":"2026-08-07T00:59:56.102596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:56.156431Z","title":"SUPERB: Speech Processing Universal PERformance Benchmark,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:56.156431Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:4956f0ef5e1c55f1dbdb2d4f024738a618b4b6d182a7096b87515c3fd23f072c","observation_id":"336466e3-0897-4f1e-822f-4945ea164596","resolution":{"observed_at":"2026-08-07T00:59:56.156431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:57.005769Z","title":"Music source separation with band-split rnn,","venue":null,"work_id":"ca8f2049-8772-4c04-8805-11c2508454f1","year":1901},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:56.215030Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:a276a62a3bef28d4ff5c4e26625cfd72eabc047271deeedfc7fcf2dcb5a5776d","observation_id":"5a8b846c-7665-4d28-965c-27e0a789b056","resolution":{"observed_at":"2026-08-07T00:59:57.051037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:56.261770Z","title":"Towards deep learning models resistant to adversarial attacks,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:56.261770Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:04ac0df1b3a4d4d20ebba8db9a0229ca34325e9ebef65d2462f6b1065298d6a8","observation_id":"6fdeea55-fb72-4c5b-b4ea-702eb9a77ec6","resolution":{"observed_at":"2026-08-07T00:59:56.261770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:59:56.909646Z","title":"Adversarial attacks on automatic speech recognition (asr): A survey,","venue":null,"work_id":"5c3d8848-0861-487b-8991-7bb36f0b08c9","year":2024},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:56.303718Z"},"links":{"citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:42a3704ff0ac84c40c886dc25d9201968caf8771a42697f463e09e84a46bd53c","observation_id":"7e6c78d8-7051-49ec-9026-26acd3bf90bd","resolution":{"observed_at":"2026-08-07T00:59:56.935708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03715","last_updated":"2026-04-24T06:25:24Z","snapshot_observed_at":"2026-07-06T19:46:02.181982Z","submitted_at":"2024-11-06T07:29:28Z","title":"MOS-Bench: Benchmarking Generalization Abilities of Subjective Speech Quality Assessment Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03715","snapshot_observed_at":"2026-08-07T00:59:55.909879Z","title":"Available: https://arxiv.org/abs/2411.03715","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T00:59:55.909879Z"},"links":{"cited_paper":"/paper/2411.03715","citing_paper":"/paper/2506.12260"},"observation_digest":"sha256:fc27a981892473abf50a8f57ae1dd6c0db0ebacc1cfed1519683ae56ba1763a7","observation_id":"0ec3ff65-a3be-4a88-863f-a0c4a7c59ef8","resolution":{"observed_at":"2026-08-07T00:59:55.909879Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.12260","last_updated":"2025-08-22T06:43:05Z","latest_version":2,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-07T00:52:24.011528Z","submitted_at":"2025-06-13T22:22:19Z","title":"Improving Speech Enhancement with Multi-Metric Supervision from Learned Quality Assessment"},"reference_resolution":{"displayed":54,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":10,"verified_exact":4,"verified_fuzzy":40},"total_outbound_references":54},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 54 of 54 outbound references and 1 inbound Pith citation observation for arXiv:2506.12260."}