{"as_of":"2026-08-08T22:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c4d348a47a00e1a7404f93a07351e3f837a7a7a1c253f398fd0719a73f71c796","coverage":[{"denominator":50,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":50,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T19:43:00.184577Z","state":"measured"},{"denominator":50,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":50,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.04845/citation-record","integrity":"/paper/2507.04845/integrity","json":"/paper/2507.04845/citation-record.json","paper":"/paper/2507.04845"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"records/1555977","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.371229Z","title":"Alpha” α and “Beta","venue":null,"work_id":"cd3d3e46-706f-46bc-aeca-d5dd029a0a1e","year":2019},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.031707Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:df7b4b39a6d636108e103d7f73e3da3831cac8bf42cc9c8c188ab10aeb1f27f6","observation_id":"5a7653ba-d2fc-4eb8-8596-656f5a9231eb","resolution":{"observed_at":"2026-08-06T19:43:00.375509Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.745897Z","title":"Alpha” ∈ RTα×dk is combined with another modality “Beta","venue":null,"work_id":"fa262b07-b094-49d4-8dfb-73a9da407ace","year":null},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.035565Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:ba708f05924840c08bba8d265881a40a767babd9299735cf5441540f21556f04","observation_id":"10b35c16-0870-4261-9298-ce3d3db7658a","resolution":{"observed_at":"2026-08-06T19:43:00.749121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.728836Z","title":"So, we adopted the inter-channel level difference (ILD) as the primary spatial feature for the SELD encoder, alongside log mel spectrograms computed independently from each channel","venue":null,"work_id":"8ab4de83-700f-4152-bdfb-0c7c80ebafa8","year":null},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.042917Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:04c937c8f4f518d66f32417715659085a571f0ede1b91bbb82a2b14dcabbc633","observation_id":"83d0c384-8d0e-4329-8e08-898ae9d40193","resolution":{"observed_at":"2026-08-06T19:43:00.731958Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.720315Z","title":"Knock” and “Bell","venue":null,"work_id":"a531a770-9cf5-4da0-8a20-f68a619c0676","year":null},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.046019Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:ebdc648680f89c9c35dd6a45acf90ffa37e334e4e0e38049c3bb818ec1d0d5dd","observation_id":"65dc058f-e69c-4d68-b656-20342f104d57","resolution":{"observed_at":"2026-08-06T19:43:00.723311Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.711290Z","title":"off-screen","venue":null,"work_id":"69b454df-6127-4c3d-a75f-f91de6ec9e75","year":2025},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.049821Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:6ddec4e0258d645e467ed3064a6bf6aa676e9a611d29c3f60f123c9972f58ebe","observation_id":"d1e8fb21-11e9-487f-8017-b3b7365235b2","resolution":{"observed_at":"2026-08-06T19:43:00.714267Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.702275Z","title":"Our approach leverages a model that integrates semantically rich feature embeddings from CLAP and OWL-ViT, fused through an adapted Conformer architecture","venue":null,"work_id":"2000a062-f978-4224-971c-bfaa62cd75de","year":2025},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.053495Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:a9baccb390836076dacb57c3db2cbb2f5dd732ca1aaf3541bdf4c299e8773b29","observation_id":"1b64737d-b1f6-4927-bdcf-4af7c361af45","resolution":{"observed_at":"2026-08-06T19:43:00.705504Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.638530Z","title":"A dataset of dynamic reverberant sound scenes with directional interferers for sound event localization and detection,","venue":null,"work_id":"7314aad4-26a0-4f37-aa2b-4a1a0b8bd943","year":2021},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.074924Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:73259299332c23a5153b5d60ad1c50572ef684c223c0aa015d6c576f7382133d","observation_id":"fc6b3b77-f741-4906-bf21-af82a4b5a976","resolution":{"observed_at":"2026-08-06T19:43:00.641911Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.693832Z","title":"Sound event localization and detection of over- lapping sources using convolutional recurrent neural networks,","venue":null,"work_id":"14af17d5-5d3c-4f17-b48f-74c7f827eb3b","year":2019},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.056268Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:71b44d9ec09bbb02433554f89827bd0ae414e23492ae05e32a865ce2c88ea70a","observation_id":"f2bb4d1e-ba8c-43a4-ba61-f77cc13ec2c5","resolution":{"observed_at":"2026-08-06T19:43:00.697013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.684112Z","title":"Sound event detection using spatial features and convolutional recurrent neural network,","venue":null,"work_id":"7cc51aed-4d36-4592-951f-76f99b9a069d","year":2017},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.059111Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:37ca55e32ecab0e7cadc21faa9d2b565492f037d55cda8aea4eaf845e5947a0b","observation_id":"c0909112-3edc-4da9-bcd7-e4b10786164b","resolution":{"observed_at":"2026-08-06T19:43:00.688103Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.675512Z","title":"Direction of arrival estima- tion for multiple sound sources using convolutional recurrent neural network,","venue":null,"work_id":"b56bdf5a-52e7-4144-8414-caf9429b7937","year":2018},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.062926Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:5c7a66a04dbfd2c0c087bcc523cc467ae5b03e940acad09e979b39be8cdf0da7","observation_id":"ac1f2b21-fe6d-42a7-9ee4-26507c174c94","resolution":{"observed_at":"2026-08-06T19:43:00.678456Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.665927Z","title":"Audio-visual cross-attention network for robotic speaker tracking,","venue":null,"work_id":"6090f467-a248-4992-98e5-2870e040de4a","year":2023},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.065774Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:f318dc6b92752b37a0160810afcbba6f0326d8f827ecf6754e2283278f38d02c","observation_id":"c34b1ae6-2317-4aad-920f-addef3746d82","resolution":{"observed_at":"2026-08-06T19:43:00.669262Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.657288Z","title":"ForecasterFlexOBM: A multi-view audio-visual dataset for flexible object-based media production,","venue":null,"work_id":"f9f0ec25-649d-405a-a301-f994ed233a9f","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.068451Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:fcb91b487abd7f72b404489cf0cc144be32de50f4d9538fb41ee4d2cb3d028a0","observation_id":"8d9b70fb-5c78-4a22-9a82-08d1321ac6bb","resolution":{"observed_at":"2026-08-06T19:43:00.660887Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.648202Z","title":"A dataset of reverberant spatial sound scenes with moving sources for sound event localization and detection,","venue":null,"work_id":"943c6a96-9aba-4b9a-bd08-770732782bfd","year":2020},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.072104Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:3d25accf24952fc0334db320f5c09e4ad6f0d388880f609c94237c5cd59238b0","observation_id":"56cd3fcb-8c7b-468f-b956-a8dcf8bc5cd7","resolution":{"observed_at":"2026-08-06T19:43:00.651703Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.572716Z","title":"Simple open-vocabulary object detection,","venue":null,"work_id":"ecbeb797-1590-43ae-a95f-581104008acb","year":2022},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.094582Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:8eb027cafa83f719bce4fad176378547486733bb85e1d43f9789ac0db8059252","observation_id":"f2f879f9-5997-4a47-9940-6c70b2015ae4","resolution":{"observed_at":"2026-08-06T19:43:00.576894Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.629643Z","title":"STARSS23: An audio-visual dataset of spatial recordings of real scenes with spatiotemporal annotations of sound events,","venue":null,"work_id":"0381b012-53f0-4131-b659-67a505177546","year":2023},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.077587Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:998bb0d055a2cbb356eb1aa65c46c56788efe3e3b4dcdcf12ea97912b86782aa","observation_id":"aa2dd003-c222-4785-93bc-71f39380c1c0","resolution":{"observed_at":"2026-08-06T19:43:00.632898Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.620001Z","title":"Baseline models and evaluation of sound event localization and detection with distance estimation in DCASE 2024 Challenge,","venue":null,"work_id":"64db25be-477a-4d9f-9199-8d401e24c5c8","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.080652Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:1b3c947dadbc03ccfcec73ad228f4efc9e1457e8b5763807d9a0b60ee3d86b8f","observation_id":"28c6262a-6740-43a5-9acb-ad5100ff0ccf","resolution":{"observed_at":"2026-08-06T19:43:00.623965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.611280Z","title":"Language models are few-shot learners,","venue":null,"work_id":"52376973-aaa3-4909-a0f2-c8a0d28a9d15","year":2020},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.083383Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:dd807cc26304d5deedffe87b0c1b646f8080101fe5eb3cc8d1ad273ab57258c3","observation_id":"f2c1a551-b1ca-4cbe-9c21-42265ee38c68","resolution":{"observed_at":"2026-08-06T19:43:00.614165Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.602444Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":"2191431d-16a2-4b25-9ca2-a311efdca320","year":2021},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.086063Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:cf6af06a8076dd123f3fa61e3bff7cf6d149212e3ca8180cda3b3c67f72accca","observation_id":"6aa9b7e1-5188-4b80-8ea1-2d1bee94342e","resolution":{"observed_at":"2026-08-06T19:43:00.605823Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.593223Z","title":"AudioGPT: understanding and generating speech, music, sound, and talking head,","venue":null,"work_id":"68ae83ba-8c6b-4d87-9549-eeb24487c387","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.088722Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:d0cf318b202a4437463ff537f0286ebb54090bfa45012a416fb3269db66f27e9","observation_id":"3d9860d6-e136-4139-af3e-7bbea2365998","resolution":{"observed_at":"2026-08-06T19:43:00.596543Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.582865Z","title":"Large-scale contrastive language-audio pretraining with feature fusion and keyword-to-caption augmentation,","venue":null,"work_id":"c1b9d362-eff7-4022-adbe-31cbdbc04565","year":2023},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.091720Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:98f01963d61ac548032a455c227875900b1f44458e21b61ef6019d1d9b42a8da","observation_id":"4b415a27-cc9e-4c4e-bc95-5728b8ad12bb","resolution":{"observed_at":"2026-08-06T19:43:00.586512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.517477Z","title":"Multi-ACCDOA: Localizing and detecting over- lapping sounds from the same class with auxiliary duplicating permu- tation invariant training,","venue":null,"work_id":"f83fddc2-5142-4ae1-b666-7beea7b97abf","year":2022},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.117153Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:ae9d19735f6781fb03f8a58c320f9ee523fd2748f910ac042018dd79b3e8a172","observation_id":"ad210db2-3e38-4bc7-a407-ae10db5a2fc7","resolution":{"observed_at":"2026-08-06T19:43:00.520694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.562320Z","title":"A four-stage data augmentation approach to resnet- conformer based acoustic modeling for sound event localization and detection,","venue":null,"work_id":"aade8191-8b56-48fd-af06-52f903980119","year":2023},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.098194Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:5ada80b471b9ff6b791ca39fd97517042e51b166546904c1c35549752daa1dcb","observation_id":"ba1d2016-0e97-4ac2-9070-64b12ec35126","resolution":{"observed_at":"2026-08-06T19:43:00.565723Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.551298Z","title":"Fusion of audio and visual embeddings for sound event localization and detection,","venue":null,"work_id":"f93dba78-d9fc-4693-a5a8-092297b58d7e","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.101728Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:1b85944b8696bb50ad3745e8bb32c086d0645f094cd49c0085a78a7c14de84d2","observation_id":"44c0f989-59bd-4cad-bb15-633281ad3d8f","resolution":{"observed_at":"2026-08-06T19:43:00.555539Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.540760Z","title":"Resnet-conformer network using multi-scale channel attention for sound event localization and detection in real scenes,","venue":null,"work_id":"215bd78d-7ea1-419e-a223-9045e4910150","year":2023},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.104342Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:afe9949eb67f1704c35a0377ac407dc2b5c7f92f305a7a3d05bbcc3f5a6510ce","observation_id":"0546a630-742c-4ef4-8a65-95633ff17233","resolution":{"observed_at":"2026-08-06T19:43:00.544602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.107548Z","title":"Conformer: Convolution-augmented transformer for speech recognition,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.107548Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:fc946a520ffe2dd77d7b37bdb91061d46f6c913ca72b1f43cc8dedccd767a939","observation_id":"f8965118-2192-4ba6-820e-feb8a0aecda4","resolution":{"observed_at":"2026-08-06T19:43:00.107548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.08644","last_updated":"2026-04-20T08:35:46Z","snapshot_observed_at":"2026-07-06T21:07:56.643874Z","submitted_at":"2025-04-11T15:43:13Z","title":"Reverberation-based Features for Sound Event Localization and Detection with Distance Estimation","version":2},"cited_work":{"arxiv_id":"2504.08644","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.08644","snapshot_observed_at":"2026-08-06T19:43:00.262298Z","title":"Reverberation-based Features for Sound Event Localization and Detection with Distance Estimation","venue":"eess.AS","work_id":"e8efb224-4f68-48d0-9c27-9111082df40f","year":2025},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.110875Z"},"links":{"cited_paper":"/paper/2504.08644","citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:50c44f811353d3e42ff3ab54606bceffae6ffda4a571d5bc0c2af0e825adea79","observation_id":"05aaa5a4-8fd6-4f33-81b2-fc7eb91b981b","resolution":{"observed_at":"2026-08-06T19:43:00.265610Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.737072Z","title":"Yet, we believe that this alternative sacrifices seman- tic richness, as pooling across channels degrades the learned feature representations","venue":null,"work_id":"31e31cad-041b-4eb1-8374-89a9321e8311","year":2025},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.039632Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:442dc810e7666286d911e17c58b246615591c85c504664bdb8487a919c0fc7ba","observation_id":"06001994-f574-4dd5-9e89-f73a46062fa0","resolution":{"observed_at":"2026-08-06T19:43:00.740135Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.526086Z","title":"Sound event detection and localization with dis- tance estimation,","venue":null,"work_id":"6cb1d880-2328-496b-aa7c-226b31db3979","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.114566Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:42fbf4892a3847fd3fafcc02b6af37a40d8df9c435bf7565a8a7ff4597f74cc2","observation_id":"33b6d5d8-0869-43a1-a9cc-f1a2cee8de1a","resolution":{"observed_at":"2026-08-06T19:43:00.529132Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.508456Z","title":"Event-independent network for polyphonic sound event localization and detection,","venue":null,"work_id":"1d0878d2-f50d-424b-8f85-edee40a96c2a","year":2020},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.120263Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:d5c0db185d9ecbc232df1ca4f6f07bf1b2d0f2b37e2908daf7367f0584de9f1d","observation_id":"8bfbcf0a-488e-48cf-9815-40346cda19bc","resolution":{"observed_at":"2026-08-06T19:43:00.512019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.499519Z","title":"An improved event-independent network for polyphonic sound event localization and detection,","venue":null,"work_id":"5522c3e3-e70e-4626-8c08-eded0f2d8b75","year":2021},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.123514Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:ce30b681256a08ed2590676a0258ceb1fdc6bdc284e3bbd58d5637c5c4de5dcd","observation_id":"3167a9db-4b10-49c9-992c-2ca0c97da4c3","resolution":{"observed_at":"2026-08-06T19:43:00.503001Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.489778Z","title":"Batch normalization: Accelerating deep net- work training by reducing internal covariate shift,","venue":null,"work_id":"9486f544-0b67-48ab-84d5-ed94cb5492db","year":2015},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.126484Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:100d4f782947ab8f5122dd663406e03f8a3634153dbbdcea7b92c849d67c3dba","observation_id":"af64aa6c-50bd-4441-b49a-eaf426bcfe0f","resolution":{"observed_at":"2026-08-06T19:43:00.493043Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.480474Z","title":"Leveraging reverberation and visual depth cues for sound event localization and detection with distance estimation,","venue":null,"work_id":"da917ce3-280f-4756-be36-d957bddc4f79","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.129132Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:bbbae5af5803e0ed388fcf84391baa5a786c84270b2879eaf06d127aa48b9722","observation_id":"6a0054de-a54b-4033-bbda-e337668a458a","resolution":{"observed_at":"2026-08-06T19:43:00.484523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.470944Z","title":"Deep residual learning for image recognition,","venue":null,"work_id":"33dafe3a-5c40-4f68-bf60-41c0c6401f5d","year":2016},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.132240Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:42ba763fe0b47a24c2cf21da89a914b779b91b902f23710869fd901824edeba9","observation_id":"f4c2311c-66b3-4fe3-92e3-c3aa62f8adf1","resolution":{"observed_at":"2026-08-06T19:43:00.475259Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14153","last_updated":"2024-11-21T14:18:10Z","snapshot_observed_at":"2026-08-05T12:26:17.752005Z","submitted_at":"2024-11-21T14:18:10Z","title":"MVANet: Multi-Stage Video Attention Network for Sound Event Localization and Detection with Source Distance Estimation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14153","snapshot_observed_at":"2026-08-06T19:43:00.135285Z","title":"MV ANet: Multi-stage video attention network for sound event localization and detection with source distance estima- tion,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.135285Z"},"links":{"cited_paper":"/paper/2411.14153","citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:e0c33f2dd3844460b0a49b607c4d4df1b01fe770eaa26022a50bbd91542a449a","observation_id":"8c269f59-3f37-4139-85da-838595967f76","resolution":{"observed_at":"2026-08-06T19:43:00.135285Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.462439Z","title":"Spatial Scaper: A library to simulate and aug- ment soundscapes for sound event localization and detection in realis- tic rooms,","venue":null,"work_id":"e56cfb1d-8e27-4c56-b188-9cffbbc6dc28","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.139318Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:d0f1c747d706e5f2c93486ec29ba65bb91438c1537d75f8820639d91ff322c42","observation_id":"04cc7c39-2166-4e55-a44f-8947b7003981","resolution":{"observed_at":"2026-08-06T19:43:00.465931Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.453713Z","title":"FSD50K: an open dataset of human-labeled sound events,","venue":null,"work_id":"38e61d2b-ddd7-40bc-af30-6a3e04eb3422","year":2022},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.142517Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:b2b7bc415e72e49a1726fee0d948a7fc9314a46bf54b6e745d934f2f9e2d7f38","observation_id":"20406617-8f72-4335-b914-07438f7dc466","resolution":{"observed_at":"2026-08-06T19:43:00.456725Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.5281/zenodo.2635758","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.204479Z","title":"METU SPARG Eigenmike em32 Acoustic Impulse Response Dataset v0.1.0,","venue":null,"work_id":"6e37c6c5-6dcc-4ee7-9ea9-cbe4420cb5d4","year":2019},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.145300Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:db9d8cb56a6169bb9fd6191b6cf217b6d2ffb3696469c07eaea227e4715eb203","observation_id":"18230b25-9ce1-483f-9076-235d2659a590","resolution":{"observed_at":"2026-08-06T19:43:00.209421Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.11882","last_updated":"2021-11-23T13:51:17Z","snapshot_observed_at":"2026-08-02T17:18:59.605671Z","submitted_at":"2021-11-23T13:51:17Z","title":"Dataset of Spatial Room Impulse Responses in a Variable Acoustics Room for Six Degrees-of-Freedom Rendering and Analysis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.11882","snapshot_observed_at":"2026-08-06T19:43:00.148335Z","title":"Dataset of spatial room impulse responses in a variable acoustics room for six degrees-of- freedom rendering and analysis,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.148335Z"},"links":{"cited_paper":"/paper/2111.11882","citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:e478981dab4f66e94663a0f8a7d0c6c05701bb87a635c10c6ef69e6f04b4c747","observation_id":"cd0c6474-7890-4180-ba42-9a757fcc32b8","resolution":{"observed_at":"2026-08-06T19:43:00.148335Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1612.01840","last_updated":"2017-09-05T18:38:33Z","snapshot_observed_at":"2026-07-06T05:21:33.309016Z","submitted_at":"2016-12-06T14:58:59Z","title":"FMA: A Dataset For Music Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.01840","snapshot_observed_at":"2026-08-06T19:43:00.151931Z","title":"FMA: A dataset for music analysis,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.151931Z"},"links":{"cited_paper":"/paper/1612.01840","citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:c34a7f8702de73731810558a2a5f4a026052a7382d580ccd25264128dbe66ab0","observation_id":"3209eada-4973-45a4-aeda-701aaf59990a","resolution":{"observed_at":"2026-08-06T19:43:00.151931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.444115Z","title":"A dataset of higher-order ambisonic room impulse responses and 3d models measured in a room with varying furniture,","venue":null,"work_id":"232650bf-bf46-40fc-927b-83560c9489da","year":2021},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.155481Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:2c80be8065dc4941b1c9541c868f5b2435f50d9d23597df76d3386095b0b6d7c","observation_id":"4ebfa138-bec0-4575-8c44-5dd3680f18d4","resolution":{"observed_at":"2026-08-06T19:43:00.447759Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.435812Z","title":"Room impulse re- sponse dataset of a recording studio with variable wall paneling mea- sured using a 32-channel spherical microphone array and a B-format microphone array,","venue":null,"work_id":"18011ac6-49c5-47f7-9363-3e7196a1c0aa","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.158221Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:366c6e79674478cea631c89e3c9ab619bd9a63f7a6af92373b4aae13c7c0d195","observation_id":"76374e6f-e93d-4173-9cf9-21e7ab680101","resolution":{"observed_at":"2026-08-06T19:43:00.439013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.427021Z","title":"Data set: Eigenmike-DRIRs, KEMAR 45BA-BRIRs, RIRs and 360° pictures captured at five positions of a small conference room,","venue":null,"work_id":"eb455ebd-cc1a-408c-a1da-44977376b6d4","year":2019},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.160983Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:767d34254324160159776eb00ad1c4d6b9e81b5e1a05257874ee8775958315f2","observation_id":"61eb1f6e-e971-4863-bcef-f8adff76a778","resolution":{"observed_at":"2026-08-06T19:43:00.430256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.02988","last_updated":"2025-04-03T19:27:43Z","snapshot_observed_at":"2026-08-07T16:10:30.072908Z","submitted_at":"2025-04-03T19:27:43Z","title":"Generating Diverse Audio-Visual 360 Soundscapes for Sound Event Localization and Detection","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.02988","snapshot_observed_at":"2026-08-06T19:43:00.163799Z","title":"Generating di- verse audio-visual 360 soundscapes for sound event localization and detection,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.163799Z"},"links":{"cited_paper":"/paper/2504.02988","citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:b03e55bd6e0bb8e616885cde7e1d20397c44704457552b2b770c4f25ea86f5ca","observation_id":"bd94749f-6952-4e7d-b5b2-c6df168b3df1","resolution":{"observed_at":"2026-08-06T19:43:00.163799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.418385Z","title":"Flickr30k entities: Collecting region-to-phrase correspondences for richer image-to-sentence models,","venue":null,"work_id":"a9d0d7cb-8c41-4cc7-b727-5d8ca306da3a","year":2015},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.167856Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:7dbd923242a332a3b8b7334e194fcc580c602c924e00aa0d990762e46d1395dc","observation_id":"773fc60b-4f9c-45aa-9edc-134a438af119","resolution":{"observed_at":"2026-08-06T19:43:00.421430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.408798Z","title":"Robust and adaptive door operation with a mobile robot,","venue":null,"work_id":"eab234bb-6f81-4250-8fbd-dc86d2c495db","year":2021},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.170687Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:df6935cb1a9c9d90dbe545b1aa645219105dafe778185b155bb4d03d596e1a6c","observation_id":"2e79874e-6267-4817-9bf4-45bec38bb158","resolution":{"observed_at":"2026-08-06T19:43:00.412566Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02030","last_updated":"2024-12-06T11:22:17Z","snapshot_observed_at":"2026-08-07T21:11:05.544473Z","submitted_at":"2024-12-02T23:20:35Z","title":"NitroFusion: High-Fidelity Single-Step Diffusion through Dynamic Adversarial Training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02030","snapshot_observed_at":"2026-08-06T19:43:00.173223Z","title":"NitroFusion: High-fidelity single-step diffusion through dynamic adversarial training,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.173223Z"},"links":{"cited_paper":"/paper/2412.02030","citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:9ad6aa20c39bccc32cb02fb2012ca79181efe42d920d5fe886afa9632e116412","observation_id":"ab8944cd-4ebf-491c-a93b-75ad25f951a0","resolution":{"observed_at":"2026-08-06T19:43:00.173223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.399884Z","title":"360-Indoor: Towards learning real-world objects in 360° indoor equirectangular images,","venue":null,"work_id":"faba7fb5-4c1c-48f8-be21-9afb9b85c256","year":2020},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.176261Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:15c2a3f5122459b59878f76f9dfd48e3b068ba00c2800e3b45d0b54d78b688d5","observation_id":"31578a05-2ae2-4a00-84b7-60cdc714dda5","resolution":{"observed_at":"2026-08-06T19:43:00.403173Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.390325Z","title":"Exploring audio-visual information fusion for sound event localization and detection in low-resource realistic scenarios,","venue":null,"work_id":"8172af58-c70f-42c2-83aa-414352ffb383","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.178838Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:0e0b3df017c9eba749798df11f3786b23bcf61faba5db1efd4dd5781d202b29f","observation_id":"3eb7450e-8f09-421f-94e4-b9928b7e6809","resolution":{"observed_at":"2026-08-06T19:43:00.394140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17129","last_updated":"2024-01-29T06:05:23Z","snapshot_observed_at":"2026-07-06T17:22:33.187222Z","submitted_at":"2024-01-29T06:05:23Z","title":"Enhanced Sound Event Localization and Detection in Real 360-degree audio-visual soundscapes","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17129","snapshot_observed_at":"2026-08-06T19:43:00.181477Z","title":"Enhanced sound event localization and de- tection in real 360-degree audio-visual soundscapes,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.181477Z"},"links":{"cited_paper":"/paper/2401.17129","citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:4753eb31b029944702114ab38ababb76454607986e8f03eb92eccb4df73a25d6","observation_id":"62e96e29-627e-41a5-929a-ac74af12b4d5","resolution":{"observed_at":"2026-08-06T19:43:00.181477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.381891Z","title":"Ultralytics YOLO,","venue":null,"work_id":"26248ec9-7de3-4e8e-b031-61668edfa719","year":2025},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.184577Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:2bf68119e278224ce62054e0b55df24f9c8a10f5f9b5b025e0aaf4df5291310d","observation_id":"78a7b421-e9a3-415d-888f-4c7c84bafd09","resolution":{"observed_at":"2026-08-06T19:43:00.384730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-07T21:12:21.046827Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos"},"reference_resolution":{"displayed":50,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":3,"verified_fuzzy":40},"total_outbound_references":50},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 50 of 50 outbound references and 0 inbound Pith citation observations for arXiv:2507.04845."}