{"as_of":"2026-08-23T05:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:58d5377e8b78940508830405d1490e16ccfb581274c6e261ff3c97d7a2c6c5f7","coverage":[{"denominator":43,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":43,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T20:27:03.800014Z","state":"measured"},{"denominator":46,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":46,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T15:24:02.180525Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-10T20:27:04.247482Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"cited_work":{"arxiv_id":"2501.08587","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.08587","snapshot_observed_at":"2026-08-10T20:27:04.247482Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","venue":"cs.AI","work_id":"2dad5991-3b9d-41ec-ba9f-f8c8d337ac93","year":2025},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.594360Z"},"links":{"cited_paper":"/paper/2501.08587","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:3fce4ee3b5f6f92606d5ee7ed30f06499cd51d2910d51959a66367e8b998925e","observation_id":"2f7e9928-baca-466e-bc81-f6ec4e9c8ef6","resolution":{"observed_at":"2026-08-10T20:27:04.253045Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08587","snapshot_observed_at":"2026-08-01T06:28:46.124972Z","title":"Sound scene synthesis at the DCASE 2024 challenge,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21857","last_updated":"2026-07-23T22:58:24Z","snapshot_observed_at":"2026-08-19T20:02:59.479949Z","submitted_at":"2026-07-23T22:58:24Z","title":"SoundscapeAgent: Agentic Soundscape Construction for Controllable Synthesis and Scalable Audio-Language Supervision","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-01T06:28:46.124972Z"},"links":{"cited_paper":"/paper/2501.08587","citing_paper":"/paper/2607.21857"},"observation_digest":"sha256:fe08d6f8f74d41a0881e196b52ccf82ca9bcfc4cf3d1b3fc1bc1ce586d2c0edc","observation_id":"c708edc0-fc08-4182-8948-978c0913df5e","resolution":{"observed_at":"2026-08-01T06:28:46.124972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08587","snapshot_observed_at":"2026-08-15T15:24:02.180525Z","title":"Sound scene synthesis at the DCASE 2024 challenge,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00463","last_updated":"2026-08-01T06:04:34Z","snapshot_observed_at":"2026-08-19T17:03:33.367062Z","submitted_at":"2026-08-01T06:04:34Z","title":"Scene2Sound: Auditory-Grounded Soundscape Generation for 3D Gaussian Worlds","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T15:24:02.180525Z"},"links":{"cited_paper":"/paper/2501.08587","citing_paper":"/paper/2608.00463"},"observation_digest":"sha256:04b656b26cd78c6b564cbbcf68d570b27b22657f43fc2284acb3468a5661bcaf","observation_id":"58a07c9c-cbe5-4c96-96df-705333699a58","resolution":{"observed_at":"2026-08-15T15:24:02.180525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.08587/citation-record","integrity":"/paper/2501.08587/integrity","json":"/paper/2501.08587/citation-record.json","paper":"/paper/2501.08587"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.694613Z","title":null,"venue":null,"work_id":"89d5e202-c8c5-4260-9b07-62acae509035","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.583398Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:6a9cb37717e779ce6abe289ca1193cb870cc02101d1331784b9ea7e2e90b32dc","observation_id":"24fce64e-375e-45ed-bee2-ad952755a333","resolution":{"observed_at":"2026-08-10T20:27:04.699769Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.677283Z","title":"This is a more flexible setup than the category- based generation used in the last year [2]","venue":null,"work_id":"06440ddb-0d07-4460-98f1-a232241da037","year":null},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.589088Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:46352293cf1768a302adbe58e6c7ef3748d3492f0946a9d172f7333f3402398c","observation_id":"77ceb030-86f3-478e-9d32-62ed4ca956c6","resolution":{"observed_at":"2026-08-10T20:27:04.682834Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"cited_work":{"arxiv_id":"2501.08587","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.08587","snapshot_observed_at":"2026-08-10T20:27:04.247482Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","venue":"cs.AI","work_id":"2dad5991-3b9d-41ec-ba9f-f8c8d337ac93","year":2025},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.594360Z"},"links":{"cited_paper":"/paper/2501.08587","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:3fce4ee3b5f6f92606d5ee7ed30f06499cd51d2910d51959a66367e8b998925e","observation_id":"2f7e9928-baca-466e-bc81-f6ec4e9c8ef6","resolution":{"observed_at":"2026-08-10T20:27:04.253045Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.659578Z","title":"Objective Evaluation We employed the Fr ´echet Audio Distance (FAD) [6] with PANN- Wavegram-Logmel [7] embeddings as our primary objective metric","venue":null,"work_id":"fa44e4c9-d8d4-474a-af3f-5b7f835be201","year":null},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.600077Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:fda5def2d61cb910107f6e89147494e1c4cc79d6fe9c4c3d714da8d8fa5f6cce","observation_id":"c19976b5-8a2d-4213-a290-7d4f0a482791","resolution":{"observed_at":"2026-08-10T20:27:04.665436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.642959Z","title":"System Performance Table 1 summarizes the evaluation results","venue":null,"work_id":"6190e0b9-5246-4a3d-97cb-7c4ab4d998af","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.605674Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:d9be987df82db999d7bfe1b4dc2bb195098211f8ab0586cee67be33b9fd39b04","observation_id":"d022a7f9-bc37-4c29-a1d0-fd48f1a9fa16","resolution":{"observed_at":"2026-08-10T20:27:04.648589Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.622330Z","title":"First, the generative aspect of organizing this challenge has been costly and labor intensive","venue":null,"work_id":"9a6bb1c3-1717-48f8-8cf3-56ea8df37e92","year":2025},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.611429Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:42fe94b67fcd134f59efb87deaa513a758838c8e08ac8cf15f54009ef4d10f13","observation_id":"01984bcb-c90b-4f23-995f-a26ac8cb3f52","resolution":{"observed_at":"2026-08-10T20:27:04.627655Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.605722Z","title":"While the submit- ted systems demonstrated promising capabilities, the significant gap between synthetic and reference audio quality indicates substantial room for improvement","venue":null,"work_id":"743c1111-6991-451b-9e75-feab5230d8ae","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.616819Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:68b7c8091c42919b79105611ee05c1864dac590b8753080905f40f1470b32af3","observation_id":"d6d2267f-0b13-465c-9cca-5ebaddf896d5","resolution":{"observed_at":"2026-08-10T20:27:04.611139Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.589147Z","title":null,"venue":null,"work_id":"dca1c5db-f3c3-43a6-9013-4f7b39df6fd5","year":null},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.621939Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:b0b593727ff26dbbad1d9c347011f205202af4e11ec3cfb458dcd72e16f6002d","observation_id":"896b9a5b-301a-4fa0-bb13-5126a48df30a","resolution":{"observed_at":"2026-08-10T20:27:04.594740Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.10760","last_updated":"2022-07-21T21:19:07Z","snapshot_observed_at":"2026-08-16T16:44:11.416113Z","submitted_at":"2022-07-21T21:19:07Z","title":"A Proposal for Foley Sound Synthesis Challenge","version":1},"cited_work":{"arxiv_id":"2207.10760","doi":null,"metadata_source":"pith","pith_arxiv_id":"2207.10760","snapshot_observed_at":"2026-08-10T20:27:04.212556Z","title":"A Proposal for Foley Sound Synthesis Challenge","venue":"cs.SD","work_id":"40af326b-c5d5-4a00-9f75-d29e0dc4ccf5","year":2022},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.627264Z"},"links":{"cited_paper":"/paper/2207.10760","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:ff355a26e1d339ab36d392445a829a3683531386e14b4de641dbeb82ffd7b94b","observation_id":"a2bfb407-3084-446d-ae3c-33759162b60a","resolution":{"observed_at":"2026-08-10T20:27:04.219052Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.572385Z","title":"Foley sound synthesis at the dcase 2023 challenge,","venue":null,"work_id":"a6b9cd9d-e5d0-4457-9eac-a6859768a27c","year":2023},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.633090Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:0cd15b126afeef4a442c893e0c0ff8e54a7187c9e3707cc814281e2d319bfd9a","observation_id":"c0476e70-cdb1-4b4c-bc49-f867c67df3e0","resolution":{"observed_at":"2026-08-10T20:27:04.577548Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.12503","last_updated":"2023-09-09T15:27:58Z","snapshot_observed_at":"2026-08-16T15:59:22.404915Z","submitted_at":"2023-01-29T17:48:17Z","title":"AudioLDM: Text-to-Audio Generation with Latent Diffusion Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.12503","snapshot_observed_at":"2026-08-10T20:27:03.638263Z","title":"AudioLDM: Text-to-audio generation with latent diffusion models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.638263Z"},"links":{"cited_paper":"/paper/2301.12503","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:c3d4aca421be4c0766f77612418073181f16f1bbf83c8418c29a6142db66fa3d","observation_id":"3ea3db78-67ed-4cba-86e0-0e8a7cee0414","resolution":{"observed_at":"2026-08-10T20:27:03.638263Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.554864Z","title":"Audiocaps: Gen- erating captions for audios in the wild,","venue":null,"work_id":"0f4926da-b48f-41dc-b4a0-bd4d7f02eabb","year":2019},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.643353Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:9bd0b01e29b81bddf10db68f2627090ff2820f36b4f29d1bab12e8a1a6d21edc","observation_id":"fdb65355-b319-4204-a4ba-312a42b89132","resolution":{"observed_at":"2026-08-10T20:27:04.561227Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.535871Z","title":"Audio set: An ontology and human-labeled dataset for audio events,","venue":null,"work_id":"de8c3764-323b-4ac2-bba1-7fc616f7bce5","year":2017},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.648007Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:bb2e6767f9c554897260bb6654ea2e8f0ef424a03a930a36734cde615499b4cf","observation_id":"6a829246-5d5c-43f3-bdd3-6d6bfaff6bd4","resolution":{"observed_at":"2026-08-10T20:27:04.541639Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.518590Z","title":"Fr ´echet audio distance: A reference-free metric for evaluating music enhancement algorithms","venue":null,"work_id":"257b90ba-8bb6-4e7d-ae0e-eace1e6ba6fc","year":2019},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.652982Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:f1e2833ff974a1c6e00faaa0796773a2d2f629525763ef01e59bc5a2b213ebbf","observation_id":"b25d28e6-1320-4048-9135-74800fd6d976","resolution":{"observed_at":"2026-08-10T20:27:04.523789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.501084Z","title":"Panns: Large-scale pretrained audio neural net- works for audio pattern recognition,","venue":null,"work_id":"e01fbf3e-e1f3-4f0f-bf60-ed61f6cde359","year":2020},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.657324Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:b03153ea631b4e1ea6993f5710a9b8a7bf46472920964983e35bd7ded89a8a02","observation_id":"ae928ba1-6638-4f65-a538-1356c3f45ac3","resolution":{"observed_at":"2026-08-10T20:27:04.506246Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.484485Z","title":"Correlation of fr´echet audio dis- tance with human perception of environmental audio is em- bedding dependent,","venue":null,"work_id":"51e45764-93d0-4c6b-8ed2-f93543ad0e2d","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.661793Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:479c35c02f5826133f217b8401c06ce418e511fe7b6f3a5583a629c1e58ee503","observation_id":"ae6d1d63-b213-49e9-9052-efc718594ebb","resolution":{"observed_at":"2026-08-10T20:27:04.489672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.468130Z","title":"Sound scene synthesis with audioldm and tango2 for dcase 2024 task7,","venue":null,"work_id":"5361221f-f6b1-4d6d-8ae2-e32e202e3a6d","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.666375Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:83723036ec8977c4fcd41106bf4c761c036f009acf1a81d5ae4915e0070d1e40","observation_id":"65f02729-3396-4aa0-8e56-43c264ec4f13","resolution":{"observed_at":"2026-08-10T20:27:04.473310Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.450887Z","title":"Sound scene synthesis based on gan using contrastive learning and effective time-frequency swap cross attention mechanism,","venue":null,"work_id":"ed2b603c-f2d1-4b64-9a43-8118230d72d4","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.671176Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:b92b9fb14636ecd96af3d9df9bc0abcd9f3fa57542b72dcb31a77b6480a027d9","observation_id":"5dc700f2-4240-4786-b517-66c3ffcf2c77","resolution":{"observed_at":"2026-08-10T20:27:04.456868Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.433625Z","title":"Dif- fusion based sound scene synthesis for dcase challenge 2024 task 7,","venue":null,"work_id":"c2cd60dd-d3f4-4a08-8e67-7c66804e150b","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.676077Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:96fd8e7ccabe001b0bd70393a99fccb8e7051dcf95c1aede65fc21aac0656e30","observation_id":"2de08bad-89f7-49d7-bd1e-af93a9bebf24","resolution":{"observed_at":"2026-08-10T20:27:04.439364Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.416434Z","title":"Sound scene synthesis based on fine-tuned latent diffusion model for dcase challenge 2024 task 7,","venue":null,"work_id":"a988ea75-ca3a-4b92-8b85-a53f2b058884","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.680779Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:2560910c2fbdbd49dcf5c2a87858f685299985b0b447e827bde0c4bda7027e5e","observation_id":"51529f4f-4c18-46b9-a575-eae1a0b49120","resolution":{"observed_at":"2026-08-10T20:27:04.421853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.399951Z","title":"Challenge on sound scene synthesis: Evaluating text-to-audio generation,","venue":null,"work_id":"c48d1577-88a6-4958-b384-7249be98c783","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.685468Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:0a8f1d4671c7777f8a3a3823449d7a6085f67de54595f6713d16531a10532079","observation_id":"6aff4a8f-529d-4cb2-b20a-dc6f31b2ab39","resolution":{"observed_at":"2026-08-10T20:27:04.405134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.382972Z","title":"T-foley: A controllable waveform-domain diffusion model for temporal-event-guided foley sound synthesis,","venue":null,"work_id":"c24d126a-c858-4492-88da-ef4c5ebb9bf9","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.690229Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:96afcf5bd4a456abaad6024fa29c66f5cbefb0b1caa9285933529da90671a437","observation_id":"6bad9403-92a0-4be7-898a-fc683ea926f7","resolution":{"observed_at":"2026-08-10T20:27:04.388081Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.09162","last_updated":"2025-03-13T15:57:34Z","snapshot_observed_at":"2026-08-18T18:50:05.239341Z","submitted_at":"2024-09-13T19:44:24Z","title":"MambaFoley: Foley Sound Generation using Selective State-Space Models","version":2},"cited_work":{"arxiv_id":"2409.09162","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.09162","snapshot_observed_at":"2026-08-10T20:27:04.161780Z","title":"MambaFoley: Foley Sound Generation using Selective State-Space Models","venue":"eess.AS","work_id":"a07f8760-c992-47bc-9e47-c039ca185827","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.694927Z"},"links":{"cited_paper":"/paper/2409.09162","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:3e84bfcb0008371b138e553ca50390ae374f00e77d6d665683196883ac9aaad6","observation_id":"9b085ddf-f36b-4e14-8b47-932f52b08be0","resolution":{"observed_at":"2026-08-10T20:27:04.168186Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.366606Z","title":"Audio generation with multiple conditional diffu- sion model,","venue":null,"work_id":"73f07112-90be-4feb-82fb-5cfda3a6f803","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.699980Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:5d23fbe02531da7f061242146b9e4ca72e154fee86e3f5917715783c5d9baf9f","observation_id":"4792880d-b8f9-4c3e-b55c-e6a4eb4f081a","resolution":{"observed_at":"2026-08-10T20:27:04.372130Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.02869","last_updated":"2024-07-17T09:50:57Z","snapshot_observed_at":"2026-08-21T12:49:10.976244Z","submitted_at":"2024-07-03T07:33:14Z","title":"PicoAudio: Enabling Precise Timestamp and Frequency Controllability of Audio Events in Text-to-audio Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.02869","snapshot_observed_at":"2026-08-10T20:27:03.704933Z","title":"Picoaudio: En- abling precise timestamp and frequency controllability of audio events in text-to-audio generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.704933Z"},"links":{"cited_paper":"/paper/2407.02869","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:209d65e6626a7240bce211dc3c7500369bc26ed55a7ff73198d22225ffff227d","observation_id":"c442015e-633a-4f80-b227-07b2e24fd41f","resolution":{"observed_at":"2026-08-10T20:27:03.704933Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.347867Z","title":"Audioldm 2: Learn- ing holistic audio generation with self-supervised pretraining,","venue":null,"work_id":"dde0f3c6-cf99-493c-940b-f47830980328","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.710664Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:4782447f7fc3cb9d6f38dd52c004edc6083c4f6f72e4dec3a5a0ea3aa9f9064f","observation_id":"b8ad09ea-ca13-482f-b2c5-23ecf02bc56f","resolution":{"observed_at":"2026-08-10T20:27:04.353850Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.01044","last_updated":"2024-01-02T05:42:14Z","snapshot_observed_at":"2026-08-18T04:36:40.182437Z","submitted_at":"2024-01-02T05:42:14Z","title":"Auffusion: Leveraging the Power of Diffusion and Large Language Models for Text-to-Audio Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.01044","snapshot_observed_at":"2026-08-10T20:27:03.716112Z","title":"Auffusion: Leveraging the power of diffusion and large language models for text-to- audio generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.716112Z"},"links":{"cited_paper":"/paper/2401.01044","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:a8a1ecfb39cd5e758f6636261a4159b5ba4e42667921d29427f51f9ed6b4dc8a","observation_id":"310b16f1-67d9-462d-8ce4-46edf18bd4ea","resolution":{"observed_at":"2026-08-10T20:27:03.716112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.10819","last_updated":"2025-06-19T04:44:02Z","snapshot_observed_at":"2026-08-17T02:39:04.886528Z","submitted_at":"2024-09-17T01:27:28Z","title":"EzAudio: Enhancing Text-to-Audio Generation with Efficient Diffusion Transformer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.10819","snapshot_observed_at":"2026-08-10T20:27:03.721679Z","title":"Ezaudio: Enhancing text-to-audio gen- eration with efficient diffusion transformer,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.721679Z"},"links":{"cited_paper":"/paper/2409.10819","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:cb7a594a494def1b34c41c36fc461c0b2e4cf900dc0e4e36ebaed76d01a728dd","observation_id":"629502ad-7164-4f64-b0bd-b10ca7ec0d24","resolution":{"observed_at":"2026-08-10T20:27:03.721679Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.327939Z","title":"Fugatto 1: Foundational generative audio transformer opus 1,","venue":null,"work_id":"e2ffb53f-945f-4662-a115-7db02b71ef98","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.727108Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:8ebade746e6bf5d6b6d244f509f43863422b192dba4b86c8929d1c5f782874ab","observation_id":"6fce6b11-ee5f-457c-ab71-a405aeb3c26d","resolution":{"observed_at":"2026-08-10T20:27:04.334553Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.14358","last_updated":"2024-07-31T16:22:42Z","snapshot_observed_at":"2026-08-16T13:32:41.679971Z","submitted_at":"2024-07-19T14:40:23Z","title":"Stable Audio Open","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.14358","snapshot_observed_at":"2026-08-10T20:27:03.732934Z","title":"Stable audio open,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.732934Z"},"links":{"cited_paper":"/paper/2407.14358","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:4f99b57f5da39aac306719086458bdf5840e9eaccd57ba45f6eb734c80376faa","observation_id":"603edb4f-0b3c-4112-bda5-bf8f02ab24c6","resolution":{"observed_at":"2026-08-10T20:27:03.732934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15487","last_updated":"2024-07-08T20:15:33Z","snapshot_observed_at":"2026-08-16T13:42:01.810217Z","submitted_at":"2024-06-18T00:02:15Z","title":"Improving Text-To-Audio Models with Synthetic Captions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15487","snapshot_observed_at":"2026-08-10T20:27:03.738918Z","title":"Improving text- to-audio models with synthetic captions,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.738918Z"},"links":{"cited_paper":"/paper/2406.15487","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:a4ca47380275a49338b95080f12b6751044681694f4b0e85bdf1c06e0b925ddb","observation_id":"6257149f-256c-4648-b54d-2de605277656","resolution":{"observed_at":"2026-08-10T20:27:03.738918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.307601Z","title":"Syncfusion: Multi- modal onset-synchronized video-to-audio foley synthesis,","venue":null,"work_id":"399fe777-54bc-4ebb-82f1-24c2af448e75","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.744932Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:d58471a4e04c5069579826205e934e5a035f120008e7f6dc62b669cc8908a4b9","observation_id":"c0cf9b71-9bd3-40d9-b780-85d5570f34b5","resolution":{"observed_at":"2026-08-10T20:27:04.314130Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.283626Z","title":"Sonicvisionlm: Play- ing sound with vision language models,","venue":null,"work_id":"ef3e9010-087e-47b0-b871-9ca2ba7e1583","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.749590Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:7706150aa03e27763c6722836e4c2a46d740a63f7420953f2f84a29f7a4e8e36","observation_id":"04198b01-4384-4d8c-a34d-6f06b9467c6c","resolution":{"observed_at":"2026-08-10T20:27:04.289175Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:03.754175Z","title":"Video-foley: Two-stage video-to-sound generation via temporal event condition for fo- ley sound,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.754175Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:2a6e401c570da91a6b0633dd23888652dc4405be458b1e43e2b860b3c95d6c8c","observation_id":"f75205d1-a832-4e17-a211-7090fc7e4ca9","resolution":{"observed_at":"2026-08-10T20:27:03.754175Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05551","last_updated":"2024-12-26T15:23:36Z","snapshot_observed_at":"2026-08-19T12:24:58.835427Z","submitted_at":"2024-07-08T01:59:17Z","title":"Read, Watch and Scream! Sound Generation from Text and Video","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05551","snapshot_observed_at":"2026-08-10T20:27:03.758711Z","title":"Read, watch and scream! sound generation from text and video,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.758711Z"},"links":{"cited_paper":"/paper/2407.05551","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:28aee607d74735018728dc8efa7823ea9fb8e7c052f6ebf912630de43699eab7","observation_id":"0fe1a964-75ef-4ecf-808b-b314da4bc19c","resolution":{"observed_at":"2026-08-10T20:27:03.758711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.13720","last_updated":"2025-02-26T16:05:55Z","snapshot_observed_at":"2026-08-17T10:09:03.767186Z","submitted_at":"2024-10-17T16:22:46Z","title":"Movie Gen: A Cast of Media Foundation Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.13720","snapshot_observed_at":"2026-08-10T20:27:03.763247Z","title":"Movie gen: A cast of media foundation models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.763247Z"},"links":{"cited_paper":"/paper/2410.13720","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:670bd9e61695481ab24c63cca37fa376cece7ce304a69c81f58c1eb08c42be71","observation_id":"ae0ef7f6-0387-4fa6-b606-d8f6fae36ee0","resolution":{"observed_at":"2026-08-10T20:27:03.763247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.17698","last_updated":"2025-03-17T17:44:37Z","snapshot_observed_at":"2026-08-19T15:57:24.818033Z","submitted_at":"2024-11-26T18:59:58Z","title":"Video-Guided Foley Sound Generation with Multimodal Controls","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.17698","snapshot_observed_at":"2026-08-10T20:27:03.768169Z","title":"Video-guided foley sound generation with multimodal controls,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.768169Z"},"links":{"cited_paper":"/paper/2411.17698","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:6671ed8dd36c80cd0e63fa1e66d1411bc0ad4253250737000bd40fb248bb9a56","observation_id":"2dd8e76a-9b05-4864-af2c-d36c1d7586b2","resolution":{"observed_at":"2026-08-10T20:27:03.768169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10768","last_updated":"2024-12-14T09:36:10Z","snapshot_observed_at":"2026-08-19T19:38:43.886895Z","submitted_at":"2024-12-14T09:36:10Z","title":"VinTAGe: Joint Video and Text Conditioning for Holistic Audio Generation","version":1},"cited_work":{"arxiv_id":"2412.10768","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.10768","snapshot_observed_at":"2026-08-10T20:27:03.913459Z","title":"VinTAGe: Joint Video and Text Conditioning for Holistic Audio Generation","venue":"cs.CV","work_id":"295e9cb8-fe32-46aa-adfe-26f0070a5720","year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.773063Z"},"links":{"cited_paper":"/paper/2412.10768","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:3e536487510c038d044e430c74fa679332a993d935c63df10cc4cd0eaaa32988","observation_id":"eaa0d4c0-e44a-4942-bc74-9a9d04cb9093","resolution":{"observed_at":"2026-08-10T20:27:03.921470Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.00320","last_updated":"2025-01-04T18:12:07Z","snapshot_observed_at":"2026-08-19T20:13:28.395800Z","submitted_at":"2024-06-01T06:40:22Z","title":"Frieren: Efficient Video-to-Audio Generation Network with Rectified Flow Matching","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.00320","snapshot_observed_at":"2026-08-10T20:27:03.778565Z","title":"Frieren: Efficient video-to-audio generation with rectified flow matching,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.778565Z"},"links":{"cited_paper":"/paper/2406.00320","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:40ace654c0cd224738cb00e1ef201222e503f9fb01d8766587725eb15ce3514f","observation_id":"d656700e-3452-4743-a68e-9de0cdf232db","resolution":{"observed_at":"2026-08-10T20:27:03.778565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:27:04.265703Z","title":"Masked gener- ative video-to-audio transformers with enhanced synchronic- ity,","venue":null,"work_id":"7d2b79e4-14b0-4d9f-9914-777f7d263668","year":2025},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.783936Z"},"links":{"citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:38903f86d067d271df1c7d72a38f316524311c460adff6a31af6ff835234da6d","observation_id":"3f8e29d9-2b0f-4cb6-987b-c9d8ec0fee72","resolution":{"observed_at":"2026-08-10T20:27:04.271468Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.13689","last_updated":"2024-09-20T17:59:01Z","snapshot_observed_at":"2026-08-16T13:16:47.187765Z","submitted_at":"2024-09-20T17:59:01Z","title":"Temporally Aligned Audio for Video with Autoregression","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.13689","snapshot_observed_at":"2026-08-10T20:27:03.788957Z","title":"Temporally aligned audio for video with autoregression,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.788957Z"},"links":{"cited_paper":"/paper/2409.13689","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:7c75998d1a3eec0aee461e7b58d92823c6b4a2f2ebffac910cf5fdfec9db2f37","observation_id":"7ca429fd-5320-47ba-add0-d721791b7105","resolution":{"observed_at":"2026-08-10T20:27:03.788957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15447","last_updated":"2025-08-12T04:20:41Z","snapshot_observed_at":"2026-08-21T12:09:18.064209Z","submitted_at":"2024-11-23T04:27:19Z","title":"Gotta Hear Them All: Towards Sound Source Aware Audio Generation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15447","snapshot_observed_at":"2026-08-10T20:27:03.794177Z","title":"Gotta hear them all: Sound source aware vision to audio generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.794177Z"},"links":{"cited_paper":"/paper/2411.15447","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:eedc91b9dceda6c408924ec3209804872a27f03c7a459d8ff5a6aba96c56a31c","observation_id":"e762b05d-a28f-45fe-a7d5-0e7fd3fc0cdf","resolution":{"observed_at":"2026-08-10T20:27:03.794177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15322","last_updated":"2025-04-07T18:00:00Z","snapshot_observed_at":"2026-08-18T13:59:15.022742Z","submitted_at":"2024-12-19T18:59:55Z","title":"MMAudio: Taming Multimodal Joint Training for High-Quality Video-to-Audio Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15322","snapshot_observed_at":"2026-08-10T20:27:03.800014Z","title":"Taming multimodal joint train- ing for high-quality video-to-audio synthesis,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T20:27:03.800014Z"},"links":{"cited_paper":"/paper/2412.15322","citing_paper":"/paper/2501.08587"},"observation_digest":"sha256:d7a322c7de0a8cab2e81f07f0701d2b06199f98e5b99a5fb536027b82139b376","observation_id":"5cecb5b1-87b6-46a2-932a-2cc195f06d3d","resolution":{"observed_at":"2026-08-10T20:27:03.800014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2501.08587","last_updated":"2025-01-15T05:15:54Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-16T14:30:18.233558Z","submitted_at":"2025-01-15T05:15:54Z","title":"Sound Scene Synthesis at the DCASE 2024 Challenge"},"reference_resolution":{"displayed":43,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":16,"verified_exact":3,"verified_fuzzy":23},"total_outbound_references":43},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 43 of 43 outbound references and 3 inbound Pith citation observations for arXiv:2501.08587."}