{"as_of":"2026-08-08T11:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:19a0f4eb51fb5667a5a3f0f408e1b18982c53efcf39f8a5b52f2f756cadbbfad","coverage":[{"denominator":42,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T23:13:42.411443Z","state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2502.08939/citation-record","integrity":"/paper/2502.08939/integrity","json":"/paper/2502.08939/citation-record.json","paper":"/paper/2502.08939"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2301.02111","last_updated":"2023-01-05T15:37:15Z","snapshot_observed_at":"2026-08-07T10:11:17.796562Z","submitted_at":"2023-01-05T15:37:15Z","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.02111","snapshot_observed_at":"2026-08-07T23:13:42.173985Z","title":"Neural codec language models are zero-shot text to speech synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.173985Z"},"links":{"cited_paper":"/paper/2301.02111","citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:2235f373bc9e30ecba6c7895c45f6ffe5dca26d2bc2659ed6aa009977a1a2b3c","observation_id":"0418ca03-2808-489f-aad3-f8ec4ab9c7d3","resolution":{"observed_at":"2026-08-07T23:13:42.173985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:43.181997Z","title":"Audiogen: Textually guided audio generation,","venue":null,"work_id":"2c91290c-6b20-4d07-aca2-f9d77c5056c2","year":2023},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.180179Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:495549b3b69870d7ca1d7ac2e824c5bc061fdb7b969390d612fb8e7e77614a7e","observation_id":"a89ac15a-938f-40d2-904a-c1f6d9a9bf5f","resolution":{"observed_at":"2026-08-07T23:13:43.187496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:43.164676Z","title":"Simple and controllable music generation,","venue":null,"work_id":"05d0ffa9-f59a-4ed2-959e-2a7629c0f5a0","year":2024},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.185540Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:550a0381542c521718b09a1378532e8c90f61a91c565ee0b8d0c10a0650a1ed2","observation_id":"53e9f589-9743-4d12-b241-e20a9bbbfc89","resolution":{"observed_at":"2026-08-07T23:13:43.169826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.11325","last_updated":"2023-01-26T18:58:53Z","snapshot_observed_at":"2026-07-06T14:45:00.730733Z","submitted_at":"2023-01-26T18:58:53Z","title":"MusicLM: Generating Music From Text","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.11325","snapshot_observed_at":"2026-08-07T23:13:42.191057Z","title":"Musiclm: Generating music from text,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.191057Z"},"links":{"cited_paper":"/paper/2301.11325","citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:ccaa261d88e29cec59b9afc7ccb3b397f3a8b6ca024cb872fd271207a1336077","observation_id":"9806fd3a-59fd-4dca-8c3d-4e387c42c977","resolution":{"observed_at":"2026-08-07T23:13:42.191057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:43.147084Z","title":"Audioldm: Text-to-audio generation with latent diffusion models,","venue":null,"work_id":"0c58756a-310c-4fe2-a91c-b6490cae38ef","year":2023},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.197118Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:3a0be11cbe3b2cecd598b9054ba5827739f0d2b2039338e45e539a939d8b161d","observation_id":"1f44e39b-9a4a-4104-acbb-b4168e6c8942","resolution":{"observed_at":"2026-08-07T23:13:43.152399Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:43.129546Z","title":"Diffwave: A versatile diffusion model for audio synthesis,","venue":null,"work_id":"bc8d4245-b67b-4771-9872-54eb45895038","year":2021},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.202715Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:be37da6ed19d4c8b9e3535476f84b13fdbcf414b0a7d83c4717249ffd3666ded","observation_id":"bc287e45-075f-4af9-89e2-de71baaa5462","resolution":{"observed_at":"2026-08-07T23:13:43.134513Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.208936Z","title":"Diffusion models: A comprehensive survey of methods and applications,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.208936Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:aace732254aa3bc8a6cff69a970d05e46a6a9c92298588deb161f3d323ba99de","observation_id":"e2fdf6a0-e066-499f-bb27-69e72fc1a7cb","resolution":{"observed_at":"2026-08-07T23:13:42.208936Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.215345Z","title":"Attention is all you need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.215345Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:5650776776b36f72645b0c543f8ed518ef1d3a51730299d8d551bd5bb12d0abd","observation_id":"54c47134-2246-416b-a041-87ef0f9f2be8","resolution":{"observed_at":"2026-08-07T23:13:42.215345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:43.088716Z","title":"Soundstream: An end-to-end neural audio codec,","venue":null,"work_id":"7ccdf196-4dd9-424f-9e77-2291a01e7990","year":2021},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.220866Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:e6c9adcf0ee44fd5c3a489e99b8b822ed76aa6348ab2521a57b1c9f63bcfe764","observation_id":"d3bb818b-f8cc-47ee-a029-e5cc4d59c88d","resolution":{"observed_at":"2026-08-07T23:13:43.094405Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.226014Z","title":"High fidelity neural audio compression,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.226014Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:e86a2505a2127434a672626e8d9060a698292dc604e6f8965cbb6bbcbec29a3c","observation_id":"cf558b7c-8381-4494-b698-45d81f64d240","resolution":{"observed_at":"2026-08-07T23:13:42.226014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.231680Z","title":"High- fidelity audio compression with improved rvqgan,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.231680Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:e4e91df65a127d33c1d7d481a698e98c1dcb77c41d7bd19c24b966ed6a06f8c4","observation_id":"04ac12ca-c2c5-4bae-9f0a-49763316a423","resolution":{"observed_at":"2026-08-07T23:13:42.231680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:43.044376Z","title":"Music controlnet: Multiple time-varying controls for music generation,","venue":null,"work_id":"318ebe7c-d810-4ad0-a024-0627f676d97a","year":2024},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.238594Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:c06f1020b7574a87f64ec71a92e405c71d4ffff20daf8263c775387d4b4f67f1","observation_id":"d92987b5-6088-476e-a8b8-6de5b051b117","resolution":{"observed_at":"2026-08-07T23:13:43.049552Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.02252","last_updated":"2025-02-01T00:25:09Z","snapshot_observed_at":"2026-07-06T17:54:47.689340Z","submitted_at":"2024-04-02T19:18:42Z","title":"SMITIN: Self-Monitored Inference-Time INtervention for Generative Music Transformers","version":2},"cited_work":{"arxiv_id":"2404.02252","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.02252","snapshot_observed_at":"2026-08-07T23:13:42.524127Z","title":"SMITIN: Self-Monitored Inference-Time INtervention for Generative Music Transformers","venue":"cs.SD","work_id":"2c76a639-0cd4-417a-a30b-e8605eebc35c","year":2024},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.245684Z"},"links":{"cited_paper":"/paper/2404.02252","citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:3f83bfeac85ded2a72a8f7b315d1f0da8be8a9bedc021ee73c0d401e345f6c9e","observation_id":"0355d789-1aed-4d12-aee0-29dd76f54eed","resolution":{"observed_at":"2026-08-07T23:13:42.529823Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:43.025185Z","title":"Neural audio synthesis of musical notes with wavenet autoencoders,","venue":null,"work_id":"c58c4caf-8e6e-4321-a5da-d93432b8e128","year":2017},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.252510Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:ae2c3df1b49aea04a662680cab308ca08139d2251f7547cc37a532ae3b6dbe40","observation_id":"80123197-06b3-4632-bb5d-21f9790d6282","resolution":{"observed_at":"2026-08-07T23:13:43.030578Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:43.006740Z","title":"Gansynth: Adversarial neural audio synthesis,","venue":null,"work_id":"f3491ba1-dc70-4e15-863e-715a0b03d58c","year":2019},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.259630Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:64f1ed2e7f061863b4c8e659af3fa355641d9cc0dbd0c5df42465a087cb79783","observation_id":"2e86e79f-a8ea-4fdc-8862-a77d54e14e9b","resolution":{"observed_at":"2026-08-07T23:13:43.012638Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.988586Z","title":"DDSP: differentiable digital signal processing,","venue":null,"work_id":"cc064245-7618-4a0a-8814-b4ba33ee9ded","year":2020},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.264584Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:2d80c26691a94fc6d4e830defc4fe532590dd42aed1c64cbe004bd753d589e3d","observation_id":"26ccaa0d-f7ee-441f-8758-f6a9b11df83f","resolution":{"observed_at":"2026-08-07T23:13:42.994116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.970755Z","title":"Sing: Symbol-to-instrument neural generator,","venue":null,"work_id":"270d8190-0b4a-4267-9882-7cfbd7e64e1f","year":2018},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.271412Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:081b54028a7e00ad8de51fbcd0e8a9a5efc9b3cf0cf5fe9872e7fa79ad86d40b","observation_id":"68265b60-3bc4-420d-9c7e-ffeea03f5d9a","resolution":{"observed_at":"2026-08-07T23:13:42.976344Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.950201Z","title":"Neural music synthesis for flexible timbre control,","venue":null,"work_id":"fc9666e6-502a-49e4-9101-1905b69943d6","year":2019},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.277056Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:192b4de5b092f724306ebc00301ba489b5192c1a84b2cb78a60f1575d7bb890f","observation_id":"738779d3-b239-453c-8a37-637852ebbdd2","resolution":{"observed_at":"2026-08-07T23:13:42.956250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.931323Z","title":"Neural waveshaping synthesis,","venue":null,"work_id":"2e2f57e9-3c86-4ec9-84c9-3bb7a49c470d","year":2021},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.283637Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:cdca38a33803fe1300218f0d530b821890c92829a197028e5527f0774d1cbfe5","observation_id":"7e165995-7efd-4550-8af7-186f734afcbf","resolution":{"observed_at":"2026-08-07T23:13:42.937282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.913260Z","title":"Differen- tiable wavetable synthesis,","venue":null,"work_id":"0e93eb7e-0119-426e-9546-83b54131a9c1","year":2022},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.288630Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:cdc0e29873bea6da64df85394199f9df03795eb239df77555ebaa694288567da","observation_id":"b10ea1f9-1e7d-4120-b89b-d775ce8fc190","resolution":{"observed_at":"2026-08-07T23:13:42.919345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.896545Z","title":"DDX7: differentiable FM synthesis of musical instrument sounds,","venue":null,"work_id":"bc1ebb57-c472-446b-92e7-fb96c65da355","year":2022},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.294260Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:124fea9ef052c30e476472e361da7ccc70c423a3365f6be3c8ce80a9ff0976c6","observation_id":"9b7f4f89-8c10-4355-9a3c-91fcb6af0a56","resolution":{"observed_at":"2026-08-07T23:13:42.901523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.876491Z","title":"A review of differentiable digital signal processing for music and speech synthesis,","venue":null,"work_id":"42dcb768-39ca-45ea-a3be-bddbf2070c26","year":2024},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.300387Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:6ab0d6950a71b9203db50d180ee92822516e313128edd6beb0f0c6e23cde7776","observation_id":"d74222ba-0ec7-43c0-aae5-2450080bda5d","resolution":{"observed_at":"2026-08-07T23:13:42.882603Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.857894Z","title":"Neural music instrument cloning from few samples,","venue":null,"work_id":"829afd98-a464-4655-bbfb-ef7cf00cee20","year":2022},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.305752Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:75c07360a0d2b921696cb9125d8737670a61a60b7ec841adee0829f8e93c730c","observation_id":"d485e751-b05e-46ad-92bb-5ab9ca70042e","resolution":{"observed_at":"2026-08-07T23:13:42.863281Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.840298Z","title":"Ganstrument: Adversarial in- strument sound synthesis with pitch-invariant instance conditioning,","venue":null,"work_id":"05519cad-f564-43f9-a288-68862243b28b","year":2023},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.311270Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:754612dc785f48dd09805a0fc69fd93a6bf24bf967627f702a66ee9be306d94b","observation_id":"f86020bc-39eb-4e05-ab21-f80f82ff95cd","resolution":{"observed_at":"2026-08-07T23:13:42.845559Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04339","last_updated":"2023-11-07T20:45:59Z","snapshot_observed_at":"2026-07-06T16:44:27.404644Z","submitted_at":"2023-11-07T20:45:59Z","title":"InstrumentGen: Generating Sample-Based Musical Instruments From Text","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04339","snapshot_observed_at":"2026-08-07T23:13:42.317111Z","title":"Instrumentgen: Generating sample-based musical instruments from text,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.317111Z"},"links":{"cited_paper":"/paper/2311.04339","citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:046c8a10d2fe4e005471cc2f925e2daa2d978f2232f4827cbf3e5ae8c4cbda48","observation_id":"089807ce-d281-471b-baed-d3195fa1e3a3","resolution":{"observed_at":"2026-08-07T23:13:42.317111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15641","last_updated":"2024-07-22T13:59:58Z","snapshot_observed_at":"2026-07-06T18:50:04.824067Z","submitted_at":"2024-07-22T13:59:58Z","title":"Generating Sample-Based Musical Instruments Using Neural Audio Codec Language Models","version":1},"cited_work":{"arxiv_id":"2407.15641","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.15641","snapshot_observed_at":"2026-08-07T23:13:42.474980Z","title":"Generating Sample-Based Musical Instruments Using Neural Audio Codec Language Models","venue":"eess.AS","work_id":"00c42459-7cb2-46d5-8247-4e631673b94b","year":2024},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.322713Z"},"links":{"cited_paper":"/paper/2407.15641","citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:bc1c7c7636b20a2112560a311913ba4cc6232826d491186d7b83ed1dcf2d86aa","observation_id":"4dc71918-d20b-4010-be43-2b78a34753a0","resolution":{"observed_at":"2026-08-07T23:13:42.482847Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.821278Z","title":"Raffel, Learning-based methods for comparing sequences, with appli- cations to audio-to-midi alignment and matching","venue":null,"work_id":"0cea049b-f26e-43c9-8bc4-b0c5ff755c2e","year":2016},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.328220Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:0e6fbee82c195fcb3efa2836b7d14d6658e3f5c3f2fb0781cc1540b2c994f42d","observation_id":"bdb04870-e5ae-40c0-8287-6f93009ca468","resolution":{"observed_at":"2026-08-07T23:13:42.827237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.798717Z","title":"Show me the instruments: Musical instrument retrieval from mixture audio,","venue":null,"work_id":"496c2e33-1c7d-4c67-a58c-2863c2937a36","year":2023},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.333871Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:9f2977c0f0837d69003a687597aeb451724ea117e023c983be4ec38dc796cd26","observation_id":"a1929b24-8f7d-4cac-9e17-19f09082a614","resolution":{"observed_at":"2026-08-07T23:13:42.804071Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.780958Z","title":"Large-scale contrastive language-audio pretraining with feature fusion and keyword-to-caption augmentation,","venue":null,"work_id":"c505f390-d609-4112-a3c4-93caa6f831fe","year":2023},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.339027Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:c010131fc5aacc08ee8ffe9c9b2cc6d8c26e19f010ab9159ca635af30fd3617b","observation_id":"29f7cfcf-3487-4fa9-a7ed-a963032b5c37","resolution":{"observed_at":"2026-08-07T23:13:42.786880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.12598","last_updated":"2022-07-26T01:42:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-07-26T01:42:07Z","title":"Classifier-Free Diffusion Guidance","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.12598","snapshot_observed_at":"2026-08-07T23:13:42.343803Z","title":"Classifier-free diffusion guidance,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.343803Z"},"links":{"cited_paper":"/paper/2207.12598","citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:949e06bfa35e8f7d616c6f3d8ed0a9014c1d4247d8ff618c6806851a12ff45b1","observation_id":"f43875b9-c387-4622-b0d5-eb4c2a833444","resolution":{"observed_at":"2026-08-07T23:13:42.343803Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.350865Z","title":"Neural discrete representation learning,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.350865Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:758fe745e743e83a68f4cefc28d784a0477a0a30d9b7d55215d818509c9bec79","observation_id":"ccab8291-f85d-45f0-a5b5-fdd3981883ea","resolution":{"observed_at":"2026-08-07T23:13:42.350865Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.748395Z","title":"Advances in residual vector quantization: A review,","venue":null,"work_id":"9ac24154-2620-4ecc-b37d-8cad3a8720f1","year":1996},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.356358Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:02d3559e277eab6c483e17ba69ffa91dbf88cf79ee627f19b6e281c4675f141a","observation_id":"3e18284d-6120-48aa-9084-d1bf8e3bbc2c","resolution":{"observed_at":"2026-08-07T23:13:42.754166Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.725223Z","title":"Autoregressive image generation using residual quantization,","venue":null,"work_id":"0f72078d-0967-40be-9d62-c438aca5471a","year":2022},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.362241Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:8eb08826d235119e217a0650e6f543f618be79f7ed9f0316e08022bf693f4a37","observation_id":"f3866a45-f014-4edf-98d5-eb341ff68ec1","resolution":{"observed_at":"2026-08-07T23:13:42.731072Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.704981Z","title":"The audio/mpeg Media Type","venue":null,"work_id":"0405fe78-26b4-4402-bcd3-8b0f30beb303","year":2000},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.367320Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:a9fc408cc692b14569652440506ae5a50c047af4e4b8562cd0800010497bda17","observation_id":"0c402d97-b6e1-4d58-8325-29ed1e799882","resolution":{"observed_at":"2026-08-07T23:13:42.710802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.372042Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.372042Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:3bf0d5307862a2b5daadb2ded216e62f9d534449c9b4603f0ace232935b9e14f","observation_id":"988f029f-6b9a-4da1-bd8b-adb96a9db44d","resolution":{"observed_at":"2026-08-07T23:13:42.372042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.376590Z","title":"MT3: multi-task multitrack music transcription,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.376590Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:53e069d4bb4217ec8a5fcdc764adb9c909e8f5eb384c6b865368a10b28ffb30d","observation_id":"2eaed653-d6fe-4d5c-88ed-d6fcba9067e2","resolution":{"observed_at":"2026-08-07T23:13:42.376590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.662974Z","title":"Music mixing style transfer: A contrastive learning approach to disentangle audio effects,","venue":null,"work_id":"348a0d23-6dde-46c1-929e-cdf3b322e71b","year":2023},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.383303Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:a7610aeab08651b7fa51010aaccc8664a333b471e9483d3418e0c21d82a3c821","observation_id":"89a75ec8-ef4a-4b8d-964d-0cf6a29bead4","resolution":{"observed_at":"2026-08-07T23:13:42.668622Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.389049Z","title":"Adam: A method for stochastic optimiza- tion,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.389049Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:4ffa5132d199f5a39483c02ce9b6cdfc3be1066698f21a317d29586a82a633c2","observation_id":"f2420b84-d0b7-48c0-bfac-1c0ac1d21b32","resolution":{"observed_at":"2026-08-07T23:13:42.389049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.394380Z","title":"PyTorch Lightning,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.394380Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:99f02c20b5cb2cf136cda3cdf63ba6fdc48e5333ea7ad7fe51a5f351faed96ec","observation_id":"7ba56706-b357-48d2-a45c-cc71e784f1e4","resolution":{"observed_at":"2026-08-07T23:13:42.394380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.619305Z","title":"Mixed precision training,","venue":null,"work_id":"72abd563-3e44-4031-9a2b-e7a9a1a03cef","year":2018},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.400592Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:6d62b759c0dd8789feb5bcb5fc12ae0e64771f3262d268a86fa47bbc58dd5ca7","observation_id":"3e024c58-6e3c-4555-aab4-b9a75ce86fc5","resolution":{"observed_at":"2026-08-07T23:13:42.624789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.406086Z","title":"Mir eval: A transparent implementation of common mir metrics.,","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.406086Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:20c2b965ab17bde9ed9cd5939e84468b8f52db07af0075583a907166b7a995a2","observation_id":"4710216c-ac97-42f6-beca-83559caa5ece","resolution":{"observed_at":"2026-08-07T23:13:42.406086Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T23:13:42.590717Z","title":"Mind the gap: Understanding the modality gap in multi-modal contrastive representation learning,","venue":null,"work_id":"110e095d-139b-4856-83f7-b84a9ee65013","year":2022},"citing_paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T23:13:42.411443Z"},"links":{"citing_paper":"/paper/2502.08939"},"observation_digest":"sha256:7f98fc65ebea189c795315ba8a7148d2b155136620871880316fb7736bf09f33","observation_id":"3bf88e11-c1bb-4e5a-ab9f-ecf4da610ba6","resolution":{"observed_at":"2026-08-07T23:13:42.596187Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2502.08939","last_updated":"2025-02-13T03:40:30Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-07T23:06:47.359198Z","submitted_at":"2025-02-13T03:40:30Z","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument"},"reference_resolution":{"displayed":42,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":14,"verified_exact":2,"verified_fuzzy":26},"total_outbound_references":42},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 42 of 42 outbound references and 0 inbound Pith citation observations for arXiv:2502.08939."}