{"as_of":"2026-08-16T13:33:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8279a3c797ac582e3ad62e76230e691f603bd014bc448fd7f485ff5f1e903ed6","coverage":[{"denominator":25,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":25,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T15:33:04.909705Z","state":"measured"},{"denominator":25,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":25,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2501.13870/citation-record","integrity":"/paper/2501.13870/integrity","json":"/paper/2501.13870/citation-record.json","paper":"/paper/2501.13870"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.207871Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":null,"work_id":"7b81f39b-5a35-4025-b011-1d69b1559a0f","year":2020},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.813257Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:df082f8d6bcc8f4d534af1c721a2d386217475fafd27204f23a1a2da6af13519","observation_id":"f7e4bbcc-7c6e-41c5-9c77-639c93fb05cc","resolution":{"observed_at":"2026-08-10T15:33:05.211518Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.197602Z","title":"Hifi- gan: Generative adversarial networks for efficient and high fidelity speech synthesis","venue":null,"work_id":"05eaae19-c9f1-42ff-bdeb-b7839e904016","year":2020},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.818527Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:ff9eff5f0451949f306747a6a0b6455b99b8da84dbbaafbb79995fd64910d818","observation_id":"f992f40e-592b-40c2-8010-d9e9cf3ddb1a","resolution":{"observed_at":"2026-08-10T15:33:05.201304Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.187991Z","title":"Bigvgan: A universal neural vocoder with large-scale training","venue":null,"work_id":"94b6b9a1-9b78-45db-96cb-f8f99402c5d9","year":2022},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.822366Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:d15402e9ada787e3c34db556bcc433c6476766f740f4f43a588a2cdce0059e16","observation_id":"4c1bf083-b4ec-4bd0-a359-b1ddf3f32878","resolution":{"observed_at":"2026-08-10T15:33:05.191035Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.178281Z","title":"High-fidelity audio compression with improved rvqgan","venue":null,"work_id":"287abc08-f52c-4e8e-9244-ee2de1b7caee","year":2024},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.826444Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:3f31812a8df34c9180d0503586b33b114f8db8193578850d0256a7d6a76accff","observation_id":"4f293436-fa58-4733-b99f-df68a8e4b2df","resolution":{"observed_at":"2026-08-10T15:33:05.181645Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.167357Z","title":"Soundstream: An end-to-end neural audio codec","venue":null,"work_id":"1f3e124e-f668-48f8-8986-f3450a0a3a42","year":2021},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.830667Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:7891adbbc93c1414d656601e79777b1aa5addb9938a0af1adac160854849682a","observation_id":"ae7545fd-5955-475b-9d52-dc36c4478ec0","resolution":{"observed_at":"2026-08-10T15:33:05.170977Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.155548Z","title":"RMSSinger: Realistic-music-score based singing voice synthesis","venue":null,"work_id":"3d162cf0-4095-4463-8a25-30f609321cab","year":2023},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.834857Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:bfdc0982ed77a0b933f8edd31c15074603ec1a96a1e02ccbfbefab4a0c22c68c","observation_id":"e1d70811-52cc-4d14-ae3c-30c8ba576ab6","resolution":{"observed_at":"2026-08-10T15:33:05.159487Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.06261","last_updated":"2020-06-11T09:09:59Z","snapshot_observed_at":"2026-08-15T11:37:19.112605Z","submitted_at":"2020-06-11T09:09:59Z","title":"XiaoiceSing: A High-Quality and Integrated Singing Voice Synthesis System","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.06261","snapshot_observed_at":"2026-08-10T15:33:04.840099Z","title":"Xiaoicesing: A high-quality and integrated singing voice synthesis system","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.840099Z"},"links":{"cited_paper":"/paper/2006.06261","citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:14759ad96c5333ff2a1e22c7799dcb78fbf1a3993384ee9e6c3ce25ab67da77f","observation_id":"73f25841-d4b8-4df4-8ae0-ef2e28f59783","resolution":{"observed_at":"2026-08-10T15:33:04.840099Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.143072Z","title":"The singing voice con- version challenge 2023","venue":null,"work_id":"7d3cf743-5ccf-4966-982a-bfbbad7ed542","year":2023},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.844337Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:1028c4640a3d78c20e1cde826ea4b28cf42e68311b7d7d072af191db4e81379e","observation_id":"3d24e77e-e1d5-4003-886b-d9e53650b675","resolution":{"observed_at":"2026-08-10T15:33:05.147741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.128578Z","title":"Gr0: Self-supervised global representation learning for zero-shot voice conversion","venue":null,"work_id":"c42e2959-ab03-4c2f-86ff-c48988415eb0","year":2024},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.847862Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:131d679d48b7dfa1cfbe8bb3d491a3fe0f90cb3457d24f5a8d233c2a95631aaa","observation_id":"db7b37fb-085e-4564-91a0-0beeff7de495","resolution":{"observed_at":"2026-08-10T15:33:05.132567Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.117159Z","title":"Neural analysis and syn- thesis: Reconstructing speech from self-supervised rep- resentations","venue":null,"work_id":"65ad05ec-ae1d-4510-92b5-55f68da82df4","year":2021},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.851428Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:db92f6d5ec17069093b0e9b869f5261162a2c4f4f2d1db1efc57d8c2c760e616","observation_id":"f7ee6814-b9bb-4f6d-95fe-f69b508e0d05","resolution":{"observed_at":"2026-08-10T15:33:05.120872Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.02111","last_updated":"2023-01-05T15:37:15Z","snapshot_observed_at":"2026-08-07T10:11:17.796562Z","submitted_at":"2023-01-05T15:37:15Z","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.02111","snapshot_observed_at":"2026-08-10T15:33:04.855210Z","title":"Neural codec language models are zero-shot text to speech synthesizers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.855210Z"},"links":{"cited_paper":"/paper/2301.02111","citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:02f49067ab67527be653241f1562005aefcba8c9b850bcbd4f95fec7ecea79f0","observation_id":"8941964d-c425-4a6b-bd44-3b23717c4634","resolution":{"observed_at":"2026-08-10T15:33:04.855210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09407","last_updated":"2022-11-17T08:29:57Z","snapshot_observed_at":"2026-08-13T13:40:33.671473Z","submitted_at":"2022-11-17T08:29:57Z","title":"NANSY++: Unified Voice Synthesis with Neural Analysis and Synthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.09407","snapshot_observed_at":"2026-08-10T15:33:04.860035Z","title":"Nansy++: Unified voice synthesis with neural analysis and synthesis","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.860035Z"},"links":{"cited_paper":"/paper/2211.09407","citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:c1194a2f06f8c1a7556d84e6aebf2339cbfa9dbab577d6c4b1912268695bf8c3","observation_id":"e8e2ad1d-f0e8-43a2-84f6-02a8f55d8900","resolution":{"observed_at":"2026-08-10T15:33:04.860035Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.106155Z","title":"https://github.com/svc-develop-team/so-vits-svc","venue":null,"work_id":"1ef9755e-dfa0-484e-8cca-782a4e63e18c","year":null},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.863805Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:eccec01968d99ebe044f9bde08cf04f68d8572f058629b35e22c40a7c57ca2bf","observation_id":"bf479d61-27f8-4ea2-aeeb-c7097a6bbc52","resolution":{"observed_at":"2026-08-10T15:33:05.110350Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.095623Z","title":"Midi-voice: Expressive zero-shot singing voice synthesis via midi-driven priors","venue":null,"work_id":"a99f5f70-ea0d-4f73-8512-3fcb1199a4e3","year":2024},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.867096Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:79862d80c2fa718c4f10555bc4fb59c7e395d230aca50da46a1e29ffb3c63d56","observation_id":"d92130f6-0d7a-491d-90c8-68c397ed9c31","resolution":{"observed_at":"2026-08-10T15:33:05.099283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.083548Z","title":"Zero-shot singing voice synthesis from musical score","venue":null,"work_id":"f43e71df-04c1-4662-8592-21785350708c","year":2023},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.870646Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:55ce69521f61d6e78283f2a6ab9e939de465488ca558e1eb88370b6b6f195dc3","observation_id":"2b7caab1-5660-4e00-8121-3c59df2f1e7f","resolution":{"observed_at":"2026-08-10T15:33:05.087757Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.072244Z","title":"Expressivesinger: Multilingual and multi- style score-based singing voice synthesis with expressive performance control","venue":null,"work_id":"10442353-c3f5-484d-9a45-d9fd0693533f","year":2024},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.874394Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:ce95bcef00731c492830e386f0789c43a1c47e8fe01135f20d214fafaae28b0a","observation_id":"c1ab64ee-ba1b-4b3d-ab3f-335e3b9d454d","resolution":{"observed_at":"2026-08-10T15:33:05.075978Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.060033Z","title":"Music style transfer: A position paper","venue":null,"work_id":"78b08c92-a0a1-4140-97e5-c39011a02c82","year":2018},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.877958Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:ba4f34accf3134f9a4100d8d23214686c89170b47e82513dc9eb05b979ec9123","observation_id":"cb048025-860e-4a77-a6da-434a7b91fa0a","resolution":{"observed_at":"2026-08-10T15:33:05.064532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.047967Z","title":"A unified model for zero-shot singing voice con- version and synthesis","venue":null,"work_id":"c514787a-330f-4060-84db-18a2691bf4d1","year":2022},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.881319Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:8022bbe2cce4191921f2c271554564b11540059c7e0abd82da848b8259acbf2d","observation_id":"9ed4f525-7cb4-4ea3-9eab-157903210a62","resolution":{"observed_at":"2026-08-10T15:33:05.051902Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-15T09:40:37.844367Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-10T15:33:04.885124Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.885124Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:ac917c7f79882f8c5335637797474c33e3512e2cbb3753bbd515317037341ce1","observation_id":"05cbca40-61bf-4eb7-aefc-5ce3701351de","resolution":{"observed_at":"2026-08-10T15:33:04.885124Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.036875Z","title":"Generalized end-to-end loss for speaker verifi- cation","venue":null,"work_id":"98633486-6a9b-4c23-9774-91f438f91f2c","year":2018},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.889517Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:85029b54baf49a2029855846785f2d0d281379c2b1efeb7cd46dc26bd35b94c4","observation_id":"111cb512-15aa-4359-8f09-02b7551766ca","resolution":{"observed_at":"2026-08-10T15:33:05.040165Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.024068Z","title":"Ddsp: Differentiable digital signal processing","venue":null,"work_id":"68d6322c-0076-4cd7-a85e-8538432a47f3","year":2019},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.894802Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:e083756190c3c8eaceb08a4a4139ad3d37e4cdf20b45e06b15716bb83f0b269c","observation_id":"e69c6172-a4bb-47a2-9647-12dc649045ae","resolution":{"observed_at":"2026-08-10T15:33:05.028027Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:05.010261Z","title":"wav2vec 2.0: A framework for self- supervised learning of speech representations","venue":null,"work_id":"df209064-a738-4b4f-b68e-8a5512851087","year":2020},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.898524Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:9d6bc2d33bd314aa22dfbbf357f791a7e20d19bd62e4e60db40294d1d11c6ca0","observation_id":"79484e40-908d-4c32-b04c-47cb1807aca1","resolution":{"observed_at":"2026-08-10T15:33:05.016006Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:33:04.997926Z","title":"Dannenberg","venue":null,"work_id":"008ed55d-f619-4af5-abe7-620dcba4b46b","year":2023},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.902133Z"},"links":{"citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:fc8e94d47fe445b7de9de8a58a8b7795bc16d95c703778551b53128f1d12ba27","observation_id":"e72ca49b-5ead-41c3-a4fe-818bb6804e01","resolution":{"observed_at":"2026-08-10T15:33:05.002634Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.18802","last_updated":"2023-05-30T07:30:21Z","snapshot_observed_at":"2026-08-13T11:29:44.547086Z","submitted_at":"2023-05-30T07:30:21Z","title":"LibriTTS-R: A Restored Multi-Speaker Text-to-Speech Corpus","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.18802","snapshot_observed_at":"2026-08-10T15:33:04.905829Z","title":"Libritts-r: A restored multi-speaker text-to-speech corpus","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.905829Z"},"links":{"cited_paper":"/paper/2305.18802","citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:2ca54e6e5034513c77756d703b83206915dba5ba1168f6af18c87757d4b56472","observation_id":"c007c5c4-a1d1-4b07-821e-08a9aeac419f","resolution":{"observed_at":"2026-08-10T15:33:04.905829Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.02502","last_updated":"2022-10-05T20:19:21Z","snapshot_observed_at":"2026-08-11T15:38:14.931716Z","submitted_at":"2020-10-06T06:15:51Z","title":"Denoising Diffusion Implicit Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.02502","snapshot_observed_at":"2026-08-10T15:33:04.909705Z","title":"Denoising diffusion implicit models","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T15:33:04.909705Z"},"links":{"cited_paper":"/paper/2010.02502","citing_paper":"/paper/2501.13870"},"observation_digest":"sha256:dbe528e3b300338f7a18db63509ad78fb35eec509b8545dd7fbb3be1418dbd21","observation_id":"b2ddef1b-fd89-4250-b25c-b5f928ef472f","resolution":{"observed_at":"2026-08-10T15:33:04.909705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2501.13870","last_updated":"2025-01-23T17:41:40Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-16T01:41:06.550043Z","submitted_at":"2025-01-23T17:41:40Z","title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference"},"reference_resolution":{"displayed":25,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":6,"verified_exact":0,"verified_fuzzy":19},"total_outbound_references":25},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 25 of 25 outbound references and 0 inbound Pith citation observations for arXiv:2501.13870."}