{"as_of":"2026-08-22T01:04:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0e14dcb615215cdb901f260f60dc7554393c5aabb72328b4c2454340ce2b5ee0","coverage":[{"denominator":41,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":41,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T14:46:08.409718Z","state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T21:49:09.624117Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.20914","snapshot_observed_at":"2026-08-03T21:49:09.624117Z","title":"Learning robust spatial rep- resentations from binaural audio through feature distillation,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2511.13487","last_updated":"2026-05-30T08:37:39Z","snapshot_observed_at":"2026-08-09T14:37:38.946092Z","submitted_at":"2025-11-17T15:25:49Z","title":"Systematic Evaluation of Time-Frequency Features for Binaural Sound Source Localization","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-03T21:49:09.624117Z"},"links":{"cited_paper":"/paper/2508.20914","citing_paper":"/paper/2511.13487"},"observation_digest":"sha256:27424f8582f2517b3f37eb2c4d84dfc94da7506d40d95bdc9e837ee2faee9815","observation_id":"825c2128-d167-4878-8c1e-7f38f1205546","resolution":{"observed_at":"2026-08-03T21:49:09.624117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2508.20914/citation-record","integrity":"/paper/2508.20914/integrity","json":"/paper/2508.20914/citation-record.json","paper":"/paper/2508.20914"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:09.044832Z","title":"Mechanisms of sound localization in mammals,","venue":null,"work_id":"77bb305b-7b97-45c2-b05d-6ee6e0579bd5","year":2010},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.213415Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:ccc0ec49f5b3acc890a708728da5f73bdb29ae03d1158fcb59fd0dabbf5af777","observation_id":"e01b74be-c36e-4457-a620-329902376a41","resolution":{"observed_at":"2026-08-05T14:46:09.049927Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:09.028782Z","title":"The generalized correlation method for estimation of time delay,","venue":null,"work_id":"c373a8dd-6072-480c-aae9-de6d814dfcd4","year":1976},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.220213Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:0078764be101707d4be86c121757347725470806fb1a52658a82f3b8f31696bf","observation_id":"2991e817-7c3d-4890-8dae-64496c0728ba","resolution":{"observed_at":"2026-08-05T14:46:09.033746Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:09.014146Z","title":"Use of the crosspower-spectrum phase in acoustic event location,","venue":null,"work_id":"eeb7372b-4c8e-4c72-b1f8-3e84ae4263f7","year":1997},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.225589Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:1ddba3cd539b366025c9a7728af66dc37dca0156fcc54bc89584771044347f2e","observation_id":"0f3d1bf4-85cc-49e1-be5d-16fdc883feff","resolution":{"observed_at":"2026-08-05T14:46:09.018606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.999616Z","title":"Benesty, J","venue":null,"work_id":"28b2987c-1c92-4eed-b94c-9282fb17ae77","year":2008},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.230104Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:cabba25756896e48fefd3ae8973d22ab0928327c67708a509b37a52c9c0d77a8","observation_id":"95b7398a-45f9-41b8-b898-3808dee7056b","resolution":{"observed_at":"2026-08-05T14:46:09.004194Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.984215Z","title":"Multi-source tdoa estimation in reverberant audio using angular spectra and clustering,","venue":null,"work_id":"55071ce6-eda8-49d9-9c46-d66dd9663137","year":1950},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.234936Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:c12d1b795de6d9d426f8ddea3d19069b8f04e08983b689f969dbcca6a214c004","observation_id":"4dc0d491-3af9-478a-a429-03de3bd1fc43","resolution":{"observed_at":"2026-08-05T14:46:08.989013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.969178Z","title":"A learning-based approach to direction of arrival estimation in noisy and reverberant environments,","venue":null,"work_id":"f83e0d1e-3853-4a6d-b356-61613a10b932","year":2015},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.239633Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:c8be849d39e832dd2d32a90a4cd0b06aba64544b7ba05ef808a70fcbb1a689dc","observation_id":"8898e373-acfd-4b0e-ad12-7cc3fcf8d2be","resolution":{"observed_at":"2026-08-05T14:46:08.974027Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.952516Z","title":"Steered response power for sound source localization: a tutorial review,","venue":null,"work_id":"18478738-6b67-46a5-848e-1cd3beeedef0","year":2024},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.244907Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:1108830b9cade34d4d5c0ec67261d887b16197fc9b377e1d841511441939fefc","observation_id":"b4534029-a4e8-429c-aee8-a3ee88ea0af9","resolution":{"observed_at":"2026-08-05T14:46:08.958275Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.935441Z","title":"Multiple emitter location and signal parameter estimation,","venue":null,"work_id":"6375a532-5a32-4529-bc65-7f91391295a7","year":1986},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.249442Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:56ac68b2c733d7860362755378a69e882fdd4c3cd82af62616938b9a2b71f946","observation_id":"de0dc64f-266e-4e19-b043-84d837b2ecc4","resolution":{"observed_at":"2026-08-05T14:46:08.941149Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.920207Z","title":"Monaural sound localization,","venue":null,"work_id":"1e45e7f3-e94c-4924-bb70-9e5c752a029c","year":2011},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.254188Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:75955718fab5ba19464eff0626e5119690b9c8eb1765b4462368d7448aa4ff14","observation_id":"9d4f7036-c7c1-4dc2-a924-5fbc55966b5f","resolution":{"observed_at":"2026-08-05T14:46:08.925406Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.903837Z","title":"Exploiting deep neural networks and head movements for robust binaural localization of multiple sources in reverberant environments,","venue":null,"work_id":"0fa59e17-1a35-40c7-8e1a-ca56ea77b0c2","year":2017},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.260345Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:c6b7bd0bf20c095f207a346334b8a8825a37a0c76f94aed58b52a3f89f4569bf","observation_id":"d49a167c-bbd9-4461-9d60-536bb2ca6f44","resolution":{"observed_at":"2026-08-05T14:46:08.909423Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.888642Z","title":"Sound source localization using deep learning models,","venue":null,"work_id":"9b2833ee-3c06-4322-a4a1-f312fb3967f5","year":2017},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.265013Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:bfb6faa9c0a5bc21b1d14cc3b11276441d86a07fd43f3d818a5f6619a894bbf0","observation_id":"9530e77f-19a1-4c44-9471-0aba36018ffa","resolution":{"observed_at":"2026-08-05T14:46:08.893482Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.872928Z","title":"Deep learning based multi-source localization with source splitting and its effectiveness in multi-talker speech recognition,","venue":null,"work_id":"f4660533-b804-464d-acc4-18512aa36d5b","year":2022},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.270620Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:624729439a6825d0ec78e6f1d2fe90418ba19cd418f852519e15b5862ec32dd5","observation_id":"ed50df32-8a56-4601-85be-80ebaa235490","resolution":{"observed_at":"2026-08-05T14:46:08.878045Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.274767Z","title":"Self-supervised speech representation learning: A review,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.274767Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:3bf281baa81c5ad841cdd350964d18b9c8a43f6b577125c08edb5975c1844229","observation_id":"41acdbd7-43f4-4b71-b08a-41d8bd7fe4e7","resolution":{"observed_at":"2026-08-05T14:46:08.274767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.845969Z","title":"Sound localization based on phase difference enhancement using deep neural networks,","venue":null,"work_id":"bfec8833-252b-4ae6-b3e1-ee962f0b5271","year":2019},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.278995Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:5a1a09d6c13ab38201fc4443fd92e2f7fb473470d448f349f430a9495db4b911","observation_id":"bf3b604c-069e-43d5-912b-2900d5349c17","resolution":{"observed_at":"2026-08-05T14:46:08.851028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.829722Z","title":"Estimation reliability function assisted sound source localization with enhanced steering vector phase difference,","venue":null,"work_id":"8b130c36-89cc-4b14-a134-e25280f345d6","year":2021},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.284493Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:a69d2cef132e86f9056e57d5fc0f46fd968878585f1fcebc747d498484c25240","observation_id":"2cb8047b-e575-4dff-9daa-0b5d1394a746","resolution":{"observed_at":"2026-08-05T14:46:08.834971Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.814013Z","title":"Ipdnet: A universal direct-path ipd esti- mation network for sound source localization,","venue":null,"work_id":"61094a4f-19fd-4c96-87c6-80dd4f293f95","year":2024},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.289869Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:9b24ab815257b4f96f4dd7b211f51849392119d10bb487b1ba2f6042fac83944","observation_id":"31727c98-d854-4520-a1d1-828695d01907","resolution":{"observed_at":"2026-08-05T14:46:08.819088Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.799643Z","title":"Masked autoencoders that listen,","venue":null,"work_id":"d0f45c8b-3ddd-4dde-97a0-5392dc835494","year":2022},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.295011Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:37edb6002659172b00d9f9eae8ea3eadccc304359af39107a842b6475e9d5f37","observation_id":"6acf3416-c150-457d-92af-3d4d1313b7d4","resolution":{"observed_at":"2026-08-05T14:46:08.804121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.784484Z","title":"An unsupervised autore- gressive model for speech representation learning,","venue":null,"work_id":"baad716e-c8c6-42a5-90fa-d8e4f799cef1","year":2019},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.300736Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:91247f0087b4f8a334b904e0ea90578b2bd95144507b79f3eecd3019b67abacc","observation_id":"4e7baaa4-6e1e-40be-85fb-a8dc143221b1","resolution":{"observed_at":"2026-08-05T14:46:08.789571Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.305703Z","title":"wav2vec 2.0: A framework for self-supervised learning of speech representations,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.305703Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:199637df4912dbe078b7b45ed7528ddcd119dcfdb204b3b3dd24ed34a0d60613","observation_id":"68f24992-2b13-447c-b1f5-00ae41785ff8","resolution":{"observed_at":"2026-08-05T14:46:08.305703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.759500Z","title":"HuBERT: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":"8633c7ca-1d57-4347-8ad7-4116b454448c","year":2021},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.310472Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:9524ffd2c20bf2d2231f3f2f0d43579a564644c91531a4d1de50321d0b03de44","observation_id":"48b8179e-b7f6-48d0-9843-30fef7b7b7da","resolution":{"observed_at":"2026-08-05T14:46:08.764021Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.745374Z","title":"WavLM: Large-scale self- supervised pre-training for full stack speech processing,","venue":null,"work_id":"9d9771ab-fedd-46f7-a2f5-5cb1bd52be16","year":2021},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.315046Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:579871ee7a6c3527695aaad3eff83c7ed22b8f1150399c31dd7298be93dcf7de","observation_id":"cc46aa99-6534-46dc-ac12-ab6d6ec04d5c","resolution":{"observed_at":"2026-08-05T14:46:08.749873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.731118Z","title":"Beats: audio pre-training with acoustic tokenizers,","venue":null,"work_id":"0872b83e-bea6-4098-8356-0e73fe8f9d98","year":2023},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.320087Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:88e1687c6262ebf368cee6c7a4a7bdf1963af15e9613ddc875613ca2f2570748","observation_id":"9fac29a7-09c3-453d-862d-a760b357db23","resolution":{"observed_at":"2026-08-05T14:46:08.735583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1807.03748","last_updated":"2019-01-22T18:47:12Z","snapshot_observed_at":"2026-08-14T18:53:38.574749Z","submitted_at":"2018-07-10T16:52:11Z","title":"Representation Learning with Contrastive Predictive Coding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1807.03748","snapshot_observed_at":"2026-08-05T14:46:08.325745Z","title":"Representation learning with contrastive predictive coding,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.325745Z"},"links":{"cited_paper":"/paper/1807.03748","citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:c8a573820a4cadaf97dca626b412f028b131c859091e816a8e2f5519cf9ca532","observation_id":"1c6fbc67-efa9-451f-9fd7-651c5431b7fd","resolution":{"observed_at":"2026-08-05T14:46:08.325745Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.715610Z","title":"Towards robust speech representation learning for thousands of languages,","venue":null,"work_id":"774e42a7-8aa4-45e7-8132-6224ff10b7b8","year":2024},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.330605Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:a03bc5c7378da8c2ec580e7517236b6b2038ef4d01efd5292f650dfb16910443","observation_id":"4221496e-0c95-461a-9544-c350faae6c20","resolution":{"observed_at":"2026-08-05T14:46:08.721250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.699919Z","title":"A noise-robust self-supervised pre-training model based speech representation learning for automatic speech recognition,","venue":null,"work_id":"94fa13ef-3621-4d5a-8120-7c3583af58c9","year":2022},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.335070Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:0aaba6ca9b2f1b6e1577f111d2f201b9c8122927b9243ef90a41a0b0a2710d9b","observation_id":"2a7a3e01-ed4a-453a-ba83-012fd1a21274","resolution":{"observed_at":"2026-08-05T14:46:08.704777Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.682800Z","title":"Joint separation and localization of moving sound sources based on neural full-rank spatial covariance analysis,","venue":null,"work_id":"f0c01a96-bb52-43c1-ab2e-89fdd4721b1c","year":2023},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.339894Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:63433d44a023138f4c7efb9c39b8418822088d04ccfcb94b4dc1e2511d3a50db","observation_id":"638e40e7-fd04-403a-ac58-55bdcbeab91c","resolution":{"observed_at":"2026-08-05T14:46:08.689309Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.667603Z","title":"Unssor: Unsupervised neural speech separation by leveraging over-determined training mixtures,","venue":null,"work_id":"21844b9a-17e3-4ffe-9150-737ab82a073a","year":2023},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.345270Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:2ce3f89d7d78cee2ab222f253a379a8ac102fba55cdcd5582be9d292f28e24b9","observation_id":"e235fe6a-7151-4ab3-9089-1984e9e649fa","resolution":{"observed_at":"2026-08-05T14:46:08.672700Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.652104Z","title":"Self-supervised learning of spatial acoustic representation with cross-channel signal reconstruction and multi-channel conformer,","venue":null,"work_id":"70560c4e-240f-42ec-8307-3f972a48583b","year":2024},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.350039Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:fa6f8d1d06aee071540aff872707f1b43486a31cb1bd7b6c268ff0b2ac0fb212","observation_id":"1c95363e-3af9-4bf8-a893-f622d2038bc3","resolution":{"observed_at":"2026-08-05T14:46:08.657289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.636980Z","title":"Librispeech: An ASR corpus based on public domain audio books,","venue":null,"work_id":"06b9e200-063c-4b73-b420-e207b8ca9a45","year":2015},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.354558Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:61ace9d2d48a85eee9543fe850ecc58c35a5efdc779414761b4abf9ab500794a","observation_id":"8c35f010-987b-4638-84ef-19dcb98ac9d9","resolution":{"observed_at":"2026-08-05T14:46:08.641979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.622525Z","title":"Libri-Light: A benchmark for asr with limited or no supervision,","venue":null,"work_id":"3252d543-2899-4c75-92b1-615b31875500","year":2020},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.358869Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:571aecd1afb8c916b8c14c8e1bb63cf242fadb5c70d2427e9af743cfcbdd0522","observation_id":"ff6342d7-9d1d-43a0-a7c3-02a3f513ff9d","resolution":{"observed_at":"2026-08-05T14:46:08.627296Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.607411Z","title":"HRTF-DATABASE,","venue":null,"work_id":"afb63430-0966-4698-b7b6-1c4c0a2b10ca","year":2024},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.363215Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:b5d97138c4c35d8d862d5c63ab73926851ada22f71bcc60c6c31ba06819ce934","observation_id":"a917a2ff-8329-44e2-b3db-bae7e2e0f353","resolution":{"observed_at":"2026-08-05T14:46:08.612366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.592240Z","title":"Signal-informed dnn-based doa estimation combining an external microphone and gcc-phat features,","venue":null,"work_id":"f926a868-941b-4486-b91f-d20b1f758087","year":2022},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.367376Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:d24f24d6ee59846d95d3a3c11fad183a4466e2b00c8d0faf4f93be3c65241552","observation_id":"fc9a454c-6988-4d2b-b60f-7e41c37721b2","resolution":{"observed_at":"2026-08-05T14:46:08.597515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.576981Z","title":"Geometry-aware doa estimation using a deep neural network with mixed-data input features,","venue":null,"work_id":"98ed327e-d579-4765-96bd-1203cbb21baf","year":2023},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.371824Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:fb58b09088181a0a7417a8f2245816894e0175add8b71d50857cd1c10ebc00e0","observation_id":"f939405a-4fc7-48b7-ad18-c54ebe767224","resolution":{"observed_at":"2026-08-05T14:46:08.581661Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.562282Z","title":"Deep learning-based speech specific source localization by using binaural and monaural microphone arrays in hearing aids,","venue":null,"work_id":"0701ec94-3885-450b-9302-e52ffb2e3e9a","year":2023},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.376135Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:dd3b6f4a96f0e25e64c1661a06c7f7bde16332c9a9bcb7757ec6f3dc79d9bd4c","observation_id":"841d9b3f-f0a6-4389-9a91-8f4997672c73","resolution":{"observed_at":"2026-08-05T14:46:08.567065Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.546839Z","title":"Regression and clas- sification for direction-of-arrival estimation with convolutional recurrent neural networks,","venue":null,"work_id":"d70d1925-3c49-4f57-9953-0f610ca3bba9","year":2019},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.380514Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:97478753e8641a5c2883bb455909735063876f71dec0574e7e46d974a818d6ac","observation_id":"cc37794b-9df8-4527-ae0c-5638604aba1c","resolution":{"observed_at":"2026-08-05T14:46:08.552159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.531016Z","title":"Conformer: Convolution- augmented transformer for speech recognition,","venue":null,"work_id":"db9e86a7-6a41-4dac-aba5-59c6342af9c2","year":2020},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.386433Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:7511cfdbe95befb887177118402f1907073b50ea1d0ffa47fbf895ec20c624ff","observation_id":"fa550a59-9b7f-44d1-a1a6-3b097d7f37a6","resolution":{"observed_at":"2026-08-05T14:46:08.535995Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.515904Z","title":"The 1st clarity prediction challenge: A machine learning challenge for hearing aid intelligibility prediction","venue":null,"work_id":"3638589a-ba36-429a-aea9-8050d44f8bb8","year":2022},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.391520Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:2247a9b8b9bdc969e4b9ab394907a16218d1ad97d7e451d6fd9cbaa6e5238c4c","observation_id":"5c3b7758-e015-495c-9502-e902fe0c5c4b","resolution":{"observed_at":"2026-08-05T14:46:08.520854Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.500543Z","title":"A study on data augmentation of reverberant speech for robust speech recognition,","venue":null,"work_id":"591e6637-303f-45b3-b754-412c73680b6b","year":2017},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.396185Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:0aa8246afaa3f2f60acda597279832f190bcd5a87d51f12cc4eddd3348338e31","observation_id":"59f0fd68-4a31-477f-a247-e73d83fea68f","resolution":{"observed_at":"2026-08-05T14:46:08.505498Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.485204Z","title":"anf-generator,","venue":null,"work_id":"5d46e96b-f7ea-41ca-b0bf-52e50ea02ef5","year":2025},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.400719Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:c20f22e62b8a0bc52085cecdf4b3f00c4f23e1e3829468e1216e543dc3f0641f","observation_id":"0eb08abd-cb96-439b-bc44-a5296ac725ea","resolution":{"observed_at":"2026-08-05T14:46:08.489939Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.466772Z","title":"Speech enhancement using long short-term memory based recurrent neural networks for noise robust speaker verification,","venue":null,"work_id":"bb483cea-6aa2-4b15-9e94-826844d4bbb4","year":2016},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.405275Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:f77c6386dc57b6aab505e7ce7dec2ddd2af63d600874e030aefe075bbd94b759","observation_id":"36f12924-6059-4799-af58-4447d1e7ceee","resolution":{"observed_at":"2026-08-05T14:46:08.474288Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:46:08.409718Z","title":"Decoupled weight decay regularization,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T14:46:08.409718Z"},"links":{"citing_paper":"/paper/2508.20914"},"observation_digest":"sha256:bda63b936c7e1497e14456846d53c25b25b6b884e6574f001c05d0c8ab3880a9","observation_id":"f172ac4f-48b7-4a9f-829a-7f690a86dc77","resolution":{"observed_at":"2026-08-05T14:46:08.409718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2508.20914","last_updated":"2025-08-28T15:43:15Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-06T07:21:43.062965Z","submitted_at":"2025-08-28T15:43:15Z","title":"Learning Robust Spatial Representations from Binaural Audio through Feature Distillation"},"reference_resolution":{"displayed":41,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":4,"verified_exact":0,"verified_fuzzy":37},"total_outbound_references":41},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 22 August 2026, this Paper Citation Record lists 41 of 41 outbound references and 1 inbound Pith citation observation for arXiv:2508.20914."}