{"as_of":"2026-08-12T21:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b9152e256e389e6239f75eb5a56a28869d2486de98f48a2d81abf9006e71aa48","coverage":[{"denominator":51,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":51,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T21:55:00.437900Z","state":"measured"},{"denominator":51,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":51,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2508.07829/citation-record","integrity":"/paper/2508.07829/integrity","json":"/paper/2508.07829/citation-record.json","paper":"/paper/2508.07829"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T21:54:55.018861Z","title":"Gpt-4 technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.018861Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:7aded76d7c267b300308d1bc7d682f4f644e704efa759f520566abc08974eb90","observation_id":"d799a26c-be9e-4a90-bdc0-004f4d199ce9","resolution":{"observed_at":"2026-08-05T21:54:55.018861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-05T21:54:55.093314Z","title":"Llama 2: Open foundation and fine-tuned chat models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.093314Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:12cd264576ee8984009db792f7a526fcc349cd1ea65a7aaacb280b670a26f0b0","observation_id":"4e1cdf58-475e-45f1-b332-78498679b8c4","resolution":{"observed_at":"2026-08-05T21:54:55.093314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-08-11T01:48:59.557045Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-05T21:54:55.207940Z","title":"Deepseek-v3 technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.207940Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:b6dec2da8e809b98addb48efac8bb1595280c059609efd25ff6da440b97a808f","observation_id":"24a23040-184e-4408-8fb7-4eb7dd8f6171","resolution":{"observed_at":"2026-08-05T21:54:55.207940Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:09.594388Z","title":"Virtanen, M","venue":null,"work_id":"d0b4bf96-f7c0-44c8-ab0f-a6814f2e479d","year":2017},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.290539Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:a1dfc11ab0292e184b8ede3c3ffa1ea68203382dfe8887d14985d622ff726677","observation_id":"706dea90-ca0a-4a26-8799-cf7c8e25b31a","resolution":{"observed_at":"2026-08-05T21:55:09.763230Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:09.308624Z","title":"Sound event detection in domestic environments with weakly labeled data and soundscape synthesis,","venue":null,"work_id":"f1920bf5-0b41-47ec-8938-2369a5c41ae5","year":2019},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.368526Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:0136bc69408ef24a9a3833e5dc8e9ebe6d329514c19e36e541dd64ba3cd23feb","observation_id":"fef540cf-73ea-4132-a74f-d3f7504bc6b3","resolution":{"observed_at":"2026-08-05T21:55:09.449372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:08.996537Z","title":"Convolutional recurrent neural networks for polyphonic sound event detection,","venue":null,"work_id":"a63992ca-be84-4383-bead-1fd46ee3fe1b","year":2017},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.459643Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:bcd63f9a9dc3710626e231fc7bf46e343bbae30d1cb0a07879bb665c02758bb4","observation_id":"5bd1999d-f935-467f-9f97-9e972676f921","resolution":{"observed_at":"2026-08-05T21:55:09.138402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:08.617758Z","title":"Metrics for polyphonic sound event detection,","venue":null,"work_id":"debff7bb-316a-4114-9456-3cb80657e479","year":2016},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.545487Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:516748eb7982fccad23d14a2f6d393b7ab822fb2b79af016a0da250dd20a07d2","observation_id":"eb209ae2-e210-4867-93c9-415fbddc20b0","resolution":{"observed_at":"2026-08-05T21:55:08.824581Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:08.274968Z","title":"A framework for the robust evaluation of sound event detection,","venue":null,"work_id":"4d7b2e4a-23ad-4e41-a9aa-a43770daa170","year":2020},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.616813Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:2155e35be90ffc3a7124c042a40d68faa295560887fb479a3f7ea63b9ee1a0c8","observation_id":"230cba28-708a-4c03-b05e-054c3d08235c","resolution":{"observed_at":"2026-08-05T21:55:08.447221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.07208","last_updated":"2025-08-27T08:09:06Z","snapshot_observed_at":"2026-08-08T13:25:13.685753Z","submitted_at":"2025-02-11T03:07:03Z","title":"Towards Understanding of Frequency Dependence on Sound Event Detection","version":2},"cited_work":{"arxiv_id":"2502.07208","doi":null,"metadata_source":"pith","pith_arxiv_id":"2502.07208","snapshot_observed_at":"2026-08-05T21:55:02.086940Z","title":"Towards Understanding of Frequency Dependence on Sound Event Detection","venue":"eess.AS","work_id":"e2e70295-dd51-4328-868c-5b31e026f5a7","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.618843Z"},"links":{"cited_paper":"/paper/2502.07208","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:342f0f775ee46a0faa088fb7345ecee84f98ed91df189f3e2d61544ce72e2605","observation_id":"9aa04b05-3827-4cbe-a996-4e467a4ee220","resolution":{"observed_at":"2026-08-05T21:55:02.185813Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.20857","last_updated":"2025-02-28T08:55:20Z","snapshot_observed_at":"2026-08-07T21:09:01.773987Z","submitted_at":"2025-02-28T08:55:20Z","title":"JiTTER: Jigsaw Temporal Transformer for Event Reconstruction for Self-Supervised Sound Event Detection","version":1},"cited_work":{"arxiv_id":"2502.20857","doi":null,"metadata_source":"pith","pith_arxiv_id":"2502.20857","snapshot_observed_at":"2026-08-05T21:55:01.925774Z","title":"JiTTER: Jigsaw Temporal Transformer for Event Reconstruction for Self-Supervised Sound Event Detection","venue":"eess.AS","work_id":"adee432d-4163-4ef8-ab79-dc7ca5f391eb","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.621227Z"},"links":{"cited_paper":"/paper/2502.20857","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:02ba7bd1500e22a795e8bd9b487b6a20ae3215ffece8a36736c42d1bf6747f77","observation_id":"b63d5103-2042-45f4-a322-ae8232c20e11","resolution":{"observed_at":"2026-08-05T21:55:02.001185Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:07.980508Z","title":"SpecAugment: A Simple Data Augmentation Method for Automatic Speech Recognition,","venue":null,"work_id":"1c8c02e6-692c-450a-b035-d3922cd1b210","year":2019},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.623704Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:0ef1187b19464300923458f67fd37216fe3dac1eed224021cf7140eb810e2b26","observation_id":"5af3869f-a693-4e89-8801-49d2dc06c548","resolution":{"observed_at":"2026-08-05T21:55:08.124515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:07.715689Z","title":"Conformer: Convolution- augmented Transformer for Speech Recognition,","venue":null,"work_id":"61a3b71a-006e-442d-9318-a3389e4ca984","year":2020},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.696331Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:992013039fd6d0416123a2145c2b99b32482257cb5fc42131ce555a12cbd2cd8","observation_id":"cd7fd502-ae51-4374-af28-ea291568350d","resolution":{"observed_at":"2026-08-05T21:55:07.854378Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:07.433681Z","title":"Coherence-based phonemic analysis on the ef- fect of reverberation to practical automatic speech recognition,","venue":null,"work_id":"28b1b0d5-64c4-4ecd-9ed7-b6f7fd63852a","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.792917Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:9f05b0f041d17af35c0442f38b34c3c902735a62a85cb492d6a6228c96a33152","observation_id":"e6ce8a0c-c5a9-4e18-8add-d94a0a1e06a2","resolution":{"observed_at":"2026-08-05T21:55:07.597433Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:07.130046Z","title":"wav2vec 2.0: A framework for self-supervised learning of speech representations,","venue":null,"work_id":"0aaa0871-ab2b-472c-8468-1e852cf6c7f3","year":2020},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.856904Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:5237b390b2525ede306cc5f663742fcf0722518cd5d8144f31f7780e0c091e57","observation_id":"9e209df9-6509-4644-808f-04218b52ba55","resolution":{"observed_at":"2026-08-05T21:55:07.258853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:06.901570Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":"cc1e6e9a-c708-4f2f-9352-4cce52039ce8","year":2021},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.890183Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:9a4c41c20d94a894314f56d80b76bc30392ed5a64c2234e085f1083008f4cfea","observation_id":"8f743f9e-7593-4972-969e-e9173af92efa","resolution":{"observed_at":"2026-08-05T21:55:07.011330Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:54:56.043026Z","title":"Attentive statistics pooling for deep speaker embedding,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:56.043026Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:289e54bc2d4d63f63726fc8d782cfbedc4f32dae7f91a09157a67153b291ac73","observation_id":"60b67069-b919-4e57-8af0-b7ef5baa3df6","resolution":{"observed_at":"2026-08-05T21:54:56.043026Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:06.758910Z","title":"Exploring the encoding layer and loss function in end-to-end speaker and language recognition system,","venue":null,"work_id":"f3d30652-3146-4407-8eb0-5781de8212dc","year":2018},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:56.245420Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:7ca9489ea687e76f7543d84c3a9ee39503aee49f11f7aebf1993481b59bb67bd","observation_id":"68704b73-a938-4901-9684-5e008bac89d6","resolution":{"observed_at":"2026-08-05T21:55:06.780807Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:06.649259Z","title":"Analysis-based optimization of temporal dynamic convolutional neural network for text-independent speaker verification,","venue":null,"work_id":"1512300f-c34f-436d-8297-56747a9d0949","year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:56.417343Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:f3adfbf04b2b81260df22d36a60361684c343b7b332955889b232c9a0fa5211e","observation_id":"188f8834-4afc-4ca9-bfd0-63c267134a61","resolution":{"observed_at":"2026-08-05T21:55:06.713473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:06.538687Z","title":"Integrating fre- quency translational invariance in tdnns and frequency positional in- formation in 2d resnets to enhance speaker verification,","venue":null,"work_id":"48f62d3f-3788-4093-99e8-79cbea17f837","year":2021},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:56.581826Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:caa7d92bb78ac9a609639453f4b22ebda64a19e02fcb5c41e7bf140d38702b3b","observation_id":"e88911b6-715e-4cb2-abf6-f55cbece2c99","resolution":{"observed_at":"2026-08-05T21:55:06.589708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:06.381522Z","title":"Convolution-based channel-frequency atten- tion for text-independent speaker verification,","venue":null,"work_id":"3afefbe0-dbd8-446d-a2f4-c8cd0c60871a","year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:56.697963Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:e535f3f27f7b77fdae78c82327307622e67f744ba1f4305ded22a6f79628db16","observation_id":"f6828fbc-3eea-4ec9-858f-b650999147e0","resolution":{"observed_at":"2026-08-05T21:55:06.462490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:06.227720Z","title":"Panns: Large-scale pretrained audio neural networks for audio pattern recognition,","venue":null,"work_id":"36783fe0-2da0-4329-97da-a1deb5730e0e","year":2020},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:56.862622Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:bd7ae42df0813a6b26ac72080c169ac61ed389582649e7093eaab9dc579d7b57","observation_id":"b0868592-5570-465e-a0d1-f6f08c9e6e85","resolution":{"observed_at":"2026-08-05T21:55:06.313111Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:06.037049Z","title":"Deep learning based cough detection camera using enhanced features,","venue":null,"work_id":"2c451e9b-529b-41d6-8fd8-a798c994ec87","year":2022},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:57.030581Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:b0f791c6a09bf7a6c1e55862c380672d5883dc3ff38384ab795eec8e276ae981","observation_id":"2de542e1-234c-4a89-9208-af40a6aa3584","resolution":{"observed_at":"2026-08-05T21:55:06.135508Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:05.806716Z","title":"Real-time sound recognition system for human care robot considering custom sound events,","venue":null,"work_id":"e4eaf4f8-a38a-4a82-a9bd-007adf7354ea","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:57.193849Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:21b43b56930abe3bd60f3b5755afdf87e98238b2cbeef5bdd158a83ea02a0c37","observation_id":"d64d0547-8f1c-40a6-8018-c7acb8450811","resolution":{"observed_at":"2026-08-05T21:55:05.927205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:05.631041Z","title":"Ast: Audio spectrogram trans- former,","venue":null,"work_id":"83d3fe6e-598f-4446-8882-33726e6e6bfb","year":2021},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:57.356267Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:3a668fc198465afa39c3d8bc5ea0c2bf6ac11f5e83cfb3241688adf1981af0ff","observation_id":"f3480612-f4b5-44cb-815d-fbe705c16e93","resolution":{"observed_at":"2026-08-05T21:55:05.744457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:05.490136Z","title":"Beats: Audio pre-training with acoustic tokenizers,","venue":null,"work_id":"3eeeba9c-0ad6-4143-84a9-5e9782191805","year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:57.480436Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:602c2e1a5705caff49bd1a9fd946307be3b366008dea91b876199d21b1b94fe7","observation_id":"049bd74c-97f3-445e-8e5c-83cf7d744dfe","resolution":{"observed_at":"2026-08-05T21:55:05.563083Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:05.315692Z","title":"Overview and evaluation of sound event localization and detection in dcase 2019,","venue":null,"work_id":"2cf25465-590c-451f-bf76-dbfbb4752922","year":2019},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:57.627088Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:e4e85f4970e52884a6d9d5a58177f6b189bc5ba9660847eee146bd97cb9c50a0","observation_id":"a325f9c7-30a8-497f-9c02-cd21b79416e7","resolution":{"observed_at":"2026-08-05T21:55:05.397189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:05.143613Z","title":"STARSS22: A dataset of spatial recordings of real scenes with spa- tiotemporal annotations of sound events,","venue":null,"work_id":"ee6f6c60-ff04-4bd4-98b6-75bde9420fb2","year":2022},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:57.738938Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:208a788d4915fa76a09df53d62fba0b2c08daa5ede00ff0d7db4c556c102ccda","observation_id":"02947e24-3dad-417e-89fe-c00cdcf9eba2","resolution":{"observed_at":"2026-08-05T21:55:05.229253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:04.964032Z","title":"Data augmentation and squeeze-and-excitation network on multiple dimension for sound event localization and detection in real scenes,","venue":null,"work_id":"0dcef2bd-e777-4a2c-91d9-968028dde5d2","year":2022},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:57.899374Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:fbc8f28fc250e74c97a0a12a6298d88c4f24b6dc1583c3da6b675655fd97c15f","observation_id":"9e0b3c59-c1e2-4daa-bfd4-278b4d3cd8c5","resolution":{"observed_at":"2026-08-05T21:55:05.050564Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20530","last_updated":"2025-07-28T05:27:07Z","snapshot_observed_at":"2026-08-09T00:44:55.137782Z","submitted_at":"2025-07-28T05:27:07Z","title":"Binaural Sound Event Localization and Detection based on HRTF Cues for Humanoid Robots","version":1},"cited_work":{"arxiv_id":"2507.20530","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.20530","snapshot_observed_at":"2026-08-05T21:55:01.738106Z","title":"Binaural Sound Event Localization and Detection based on HRTF Cues for Humanoid Robots","venue":"eess.AS","work_id":"a19684ed-d865-4b12-bc42-06f159a5eec5","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.024131Z"},"links":{"cited_paper":"/paper/2507.20530","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:090c589fa14ae9713da92469ec7e983363ad2330947030e5bd0a0cb471eb852c","observation_id":"30625b5f-9cdb-4af7-b716-9cdc483e3830","resolution":{"observed_at":"2026-08-05T21:55:01.847274Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:04.793104Z","title":"Automated audio captioning with recurrent neural networks,","venue":null,"work_id":"115955b1-8091-4840-ae3b-aa9f0b3f0750","year":2017},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.170275Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:14d1c41f7116ede6117cb227b8b24aa4d78308161edeeeacf55b80c243ae86e3","observation_id":"1e23078e-4c6b-4e27-8ea9-b43acbe34508","resolution":{"observed_at":"2026-08-05T21:55:04.881036Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:04.634115Z","title":"Clotho: an audio captioning dataset,","venue":null,"work_id":"bff492f5-9e07-404c-ab73-0bf2fc1422bd","year":2020},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.286059Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:74fdc757b8d9d47100ad185bbbbe19d899a4210340ee8dbcf613c267074e4529","observation_id":"1a0665ed-0268-45ac-b880-3125acd53a1d","resolution":{"observed_at":"2026-08-05T21:55:04.676532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:04.448400Z","title":"Chatgpt caption paraphrasing and fense-based caption filtering for automated audio captioning,","venue":null,"work_id":"cbc06dc6-e05d-4911-963c-cc58abf91e1a","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.404359Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:a4102c9cd42ab6006c480330f2ca70b565a03b5fca9f88e6e8072f42eec54f64","observation_id":"77818c72-35bb-4d60-bb00-17b96f46a42f","resolution":{"observed_at":"2026-08-05T21:55:04.531602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.18638","last_updated":"2024-03-27T14:44:24Z","snapshot_observed_at":"2026-07-06T17:52:00.714976Z","submitted_at":"2024-03-27T14:44:24Z","title":"Mind the Domain Gap: a Systematic Analysis on Bioacoustic Sound Event Detection","version":1},"cited_work":{"arxiv_id":"2403.18638","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.18638","snapshot_observed_at":"2026-08-05T21:55:01.548902Z","title":"Mind the Domain Gap: a Systematic Analysis on Bioacoustic Sound Event Detection","venue":"eess.AS","work_id":"6fecad86-b2fd-474a-bd65-6a83f0f20902","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.488390Z"},"links":{"cited_paper":"/paper/2403.18638","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:c7f3e55b824aea60fe3ea275f6cec55dbf2f5c913fe670914be44a6c6e3461ba","observation_id":"556bbac1-d0bc-4baf-a758-c59642c58474","resolution":{"observed_at":"2026-08-05T21:55:01.629069Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:04.305757Z","title":"Few-shot bioacoustic event detection utilizing spectro-temporal receptive field,","venue":null,"work_id":"2282fd05-985d-4d9c-8274-d4c2ef098e0f","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.572096Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:72b109eda145c4eec2907b35c2b97588867c7f98026909d67966b33a4108e0eb","observation_id":"9633a1d8-64e0-48fd-bcfc-fabe39d98b4f","resolution":{"observed_at":"2026-08-05T21:55:04.395425Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:04.171737Z","title":"Prtfnet: Hrtf individual- ization for accurate spectral cues using a compact prtf,","venue":null,"work_id":"e3a6651a-791f-4d28-910b-350d2dd37ef8","year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.674078Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:8b600949511385baaeda58afd0a4cece8699aab1bdd4cdda3723e62be9c3b83e","observation_id":"a613f9fe-77bb-4d9b-b2a5-ca500286eb07","resolution":{"observed_at":"2026-08-05T21:55:04.245019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:03.991299Z","title":"Filteraugment: An acoustic environmental data augmentation method,","venue":null,"work_id":"2ab2deb3-d288-4223-bf2c-0cf1fa52b99b","year":2022},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.787521Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:33334c0b66984af7acde57c4c3ecace1c328f5ec93835f039a281eae0a5fbae1","observation_id":"d6f13d25-fadb-4cf4-90c1-cdf14ff29460","resolution":{"observed_at":"2026-08-05T21:55:04.073974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14817","last_updated":"2025-04-21T02:50:34Z","snapshot_observed_at":"2026-08-07T16:00:42.203123Z","submitted_at":"2025-04-21T02:50:34Z","title":"DNN based HRIRs Identification with a Continuously Rotating Speaker Array","version":1},"cited_work":{"arxiv_id":"2504.14817","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.14817","snapshot_observed_at":"2026-08-05T21:55:01.314634Z","title":"DNN based HRIRs Identification with a Continuously Rotating Speaker Array","venue":"eess.AS","work_id":"d5183c36-f9fc-4b06-b7c6-8459785720f9","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.901523Z"},"links":{"cited_paper":"/paper/2504.14817","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:3ce7e4fdc1b5f75b047709dbe0f64c77062007fb4bc4403fb1e53b2f08b7ba5b","observation_id":"369327be-28db-4ef5-baf5-9b2f03d07e45","resolution":{"observed_at":"2026-08-05T21:55:01.438903Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:03.845478Z","title":"AudioLDM: Text-to-audio generation with latent diffusion models,","venue":null,"work_id":"cd01c446-6730-492d-b9ad-f7296d2e1859","year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.988369Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:1495f1e43f59a6a8930e9600cd0bf409e392d371fddaa4a1883c929d7a353df6","observation_id":"792542b4-877c-4b7c-922c-55834e3b1dff","resolution":{"observed_at":"2026-08-05T21:55:03.901509Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:03.688357Z","title":"Audiogen: Textually guided audio generation,","venue":null,"work_id":"1adec640-b89e-48ed-bea7-31040c9fbb05","year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.102365Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:c870fa6391506a2350ac8f4854c11dc7c54b2b0a8ce9144620d9977ee17f1e20","observation_id":"ada6fa31-0d43-4223-a3dc-defab0890554","resolution":{"observed_at":"2026-08-05T21:55:03.753031Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:03.478428Z","title":"Vifs: An end-to-end variational inference for foley sound synthesis,","venue":null,"work_id":"468c9414-08b1-4101-8c41-b79a02f13f51","year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.215508Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:01499d1804742cb4aef1e7578d7b1937979dbf9dca928e2beb7d9313cc8921b0","observation_id":"d7c109d0-d67f-4729-bc6d-1c51d4b0c51b","resolution":{"observed_at":"2026-08-05T21:55:03.554341Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:03.341487Z","title":"Heavily augmented sound event detection utilizing weak predictions,","venue":null,"work_id":"caf02097-7ece-45b4-b53d-87de3b904d47","year":2021},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.349962Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:7ce9a07f792289d56c858c2e69b16d048277f33a4e459aa399add065d2ecbf9e","observation_id":"7bcbe46f-95b9-49f2-91b4-e1d3c2899ae0","resolution":{"observed_at":"2026-08-05T21:55:03.411377Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:03.193229Z","title":"Filteraugment: An acoustic environmental data augmentation method,","venue":null,"work_id":"096cd250-ff3d-49e5-a112-d65df3218627","year":2022},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.472133Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:9b9561733c44085ac33ce547687e528864b9749fcc9c25e957f0dba9f99b90d3","observation_id":"21a5f27c-e399-4348-a384-1d74fc22a414","resolution":{"observed_at":"2026-08-05T21:55:03.264445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:02.984671Z","title":"Frequency Dynamic Convolution: Frequency-Adaptive Pattern Recognition for Sound Event Detection,","venue":null,"work_id":"1ed47e4e-a41a-4b71-be56-c8cd5924667b","year":2022},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.555478Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:cc7242bd6f2ad654cbdcf5df0c6ba5e8c64dfe90a9350d46d5222b3c37ec4dfc","observation_id":"1973ca4d-8cb7-4cb9-a7b5-7759246abec0","resolution":{"observed_at":"2026-08-05T21:55:03.075348Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:02.828505Z","title":"Self training and ensembling frequency dependent networks with coarse prediction pooling and sound event bounding boxes,","venue":null,"work_id":"b13e2f6e-28a0-4401-8bcc-3fc353a5d39f","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.654198Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:10808143b7c91ba4575bd99db4ba7edf65a1128602d2ac47f30ee72a8d673426","observation_id":"ded12c34-a415-43b6-9e1d-3e0d9102ebec","resolution":{"observed_at":"2026-08-05T21:55:02.909992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:02.582196Z","title":"Diversifying and expanding frequency-adaptive convolution kernels for sound event detection,","venue":null,"work_id":"e43dd54a-9390-4390-beeb-d757ebe1723e","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.795404Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:bed583a33b7b8fcf489f205731b429e4f8c44a022802ffba962801f67333df64","observation_id":"0ae9c7db-d015-4fed-a50c-4e1f603a90e2","resolution":{"observed_at":"2026-08-05T21:55:02.747332Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13312","last_updated":"2024-09-20T02:18:47Z","snapshot_observed_at":"2026-08-12T19:51:56.892965Z","submitted_at":"2024-06-19T08:02:02Z","title":"Pushing the Limit of Sound Event Detection with Multi-Dilated Frequency Dynamic Convolution","version":3},"cited_work":{"arxiv_id":"2406.13312","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.13312","snapshot_observed_at":"2026-08-05T21:55:01.157686Z","title":"Pushing the Limit of Sound Event Detection with Multi-Dilated Frequency Dynamic Convolution","venue":"eess.AS","work_id":"ce60f310-b3b1-4276-842a-8b1e114f638d","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.894205Z"},"links":{"cited_paper":"/paper/2406.13312","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:6d7300abc3741e219f6b6bbe2608c32e5e4e701761b73630d429a24eed25f290","observation_id":"05d450ad-e4a7-4d4c-820a-035c26b9ba0f","resolution":{"observed_at":"2026-08-05T21:55:01.245826Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.12670","last_updated":"2025-04-17T06:03:43Z","snapshot_observed_at":"2026-08-08T04:18:21.741361Z","submitted_at":"2025-04-17T06:03:43Z","title":"Temporal Attention Pooling for Frequency Dynamic Convolution in Sound Event Detection","version":1},"cited_work":{"arxiv_id":"2504.12670","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.12670","snapshot_observed_at":"2026-08-05T21:55:00.961516Z","title":"Temporal Attention Pooling for Frequency Dynamic Convolution in Sound Event Detection","venue":"eess.AS","work_id":"38288718-f4f7-440a-972a-57ea6f3fb17f","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.989934Z"},"links":{"cited_paper":"/paper/2504.12670","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:e00757846e1617a1090b3cdc1d0c0b099c8d702e9bb37ea40931218b5218795e","observation_id":"72f54b94-765d-44ca-ac9e-ddc6677757af","resolution":{"observed_at":"2026-08-05T21:55:01.067562Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.12785","last_updated":"2025-06-15T09:32:16Z","snapshot_observed_at":"2026-08-07T00:40:07.029946Z","submitted_at":"2025-06-15T09:32:16Z","title":"Frequency Dynamic Convolutions for Sound Event Detection","version":1},"cited_work":{"arxiv_id":"2506.12785","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.12785","snapshot_observed_at":"2026-08-05T21:55:00.763800Z","title":"Frequency Dynamic Convolutions for Sound Event Detection","venue":"eess.AS","work_id":"dc43963d-3864-4bf9-bec7-a9e062074257","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-05T21:55:00.075447Z"},"links":{"cited_paper":"/paper/2506.12785","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:060793b8260fc79eba0bd71a489a0b37c8ac825c96b9d1fc50cbdf2278044cf4","observation_id":"773b6fc6-25d1-44e5-86c6-7f4ef1b8a7d8","resolution":{"observed_at":"2026-08-05T21:55:00.851246Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.05335","last_updated":"2025-06-08T18:21:26Z","snapshot_observed_at":"2026-08-10T11:45:36.343724Z","submitted_at":"2025-05-08T15:27:43Z","title":"FLAM: Frame-Wise Language-Audio Modeling","version":2},"cited_work":{"arxiv_id":"2505.05335","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.05335","snapshot_observed_at":"2026-08-05T21:55:00.563579Z","title":"FLAM: Frame-Wise Language-Audio Modeling","venue":"cs.SD","work_id":"547395b7-81eb-43ab-8ba3-6f1d93afb6ff","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-05T21:55:00.164281Z"},"links":{"cited_paper":"/paper/2505.05335","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:0283eb989370302f9ae0ae7b8de60adf0cc54039e24b31ca8fc93e0af2e06397","observation_id":"f3f7fd21-81b5-464d-82dc-1d6b272b7750","resolution":{"observed_at":"2026-08-05T21:55:00.654739Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:02.421355Z","title":"AudioCaps: Generating captions for audios in the wild,","venue":null,"work_id":"a4e289d4-951a-474a-a3a9-82073007ae4a","year":2019},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-05T21:55:00.299272Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:3d8ffab20243c243093dc5b4f2d6953934a40568e440acd8346e208cd2051cc6","observation_id":"2d676abf-5417-479d-85b2-bcc09366cf9a","resolution":{"observed_at":"2026-08-05T21:55:02.500837Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:02.290136Z","title":"Wavcaps: A chatgpt-assisted weakly-labelled audio captioning dataset for audio-language multimodal research,","venue":null,"work_id":"6ec90f8a-8b80-49d7-9c41-46b8f09bfd19","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-05T21:55:00.437900Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:8e9fbfcd20c0fb491a5893514b1c330ade5142da51ebe03afb27f457f20fc93d","observation_id":"476396e1-5fa4-4778-a4b8-db33cee5beeb","resolution":{"observed_at":"2026-08-05T21:55:02.344444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-07T23:23:56.713686Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound"},"reference_resolution":{"displayed":51,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":4,"verified_exact":9,"verified_fuzzy":38},"total_outbound_references":51},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 51 of 51 outbound references and 0 inbound Pith citation observations for arXiv:2508.07829."}