{"as_of":"2026-08-14T22:29:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6164b13651ea4d726a2573d843433fbc56b7dfa2afe076a8d47a23cfbd00adf7","coverage":[{"denominator":30,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T22:51:50.050472Z","state":"measured"},{"denominator":30,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":30,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.03074/citation-record","integrity":"/paper/2412.03074/integrity","json":"/paper/2412.03074/citation-record.json","paper":"/paper/2412.03074"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2203.01829","last_updated":"2022-03-01T11:15:35Z","snapshot_observed_at":"2026-08-11T12:50:25.350835Z","submitted_at":"2022-03-01T11:15:35Z","title":"A Brief Overview of Unsupervised Neural Speech Representation Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.01829","snapshot_observed_at":"2026-08-11T22:51:49.884332Z","title":"A brief overview of unsupervised neural speech representation learning,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.884332Z"},"links":{"cited_paper":"/paper/2203.01829","citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:8c97de2587d1c3c850b89b1bb521c7ff63865c613a77f3efbd770119c8ad9cf8","observation_id":"28aeab77-1755-46f0-a88a-d109bc6cc833","resolution":{"observed_at":"2026-08-11T22:51:49.884332Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.05884","last_updated":"2018-02-16T01:28:23Z","snapshot_observed_at":"2026-08-14T20:03:06.478622Z","submitted_at":"2017-12-16T00:51:40Z","title":"Natural TTS Synthesis by Conditioning WaveNet on Mel Spectrogram Predictions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.05884","snapshot_observed_at":"2026-08-11T22:51:49.890550Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.890550Z"},"links":{"cited_paper":"/paper/1712.05884","citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:2ba08f021c71e254de6fdcbee3de2fb282f9dff9a8e8d70afe565eb5eb17ff60","observation_id":"24e6162f-c0ef-409a-837a-d5076b57dfee","resolution":{"observed_at":"2026-08-11T22:51:49.890550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-14T02:11:04.009144Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-11T22:51:49.895990Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.895990Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:dc0b0833c278e25bcd7bc2bb6582e22f6007dae27236ad7b139d09480c80aa19","observation_id":"ff9c97c0-aaa2-497e-b7b9-dd602b6bb6bc","resolution":{"observed_at":"2026-08-11T22:51:49.895990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.601615Z","title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech,","venue":null,"work_id":"52b06fa6-f2d7-44dd-966d-89fd42f9c02d","year":2021},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.901188Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:ee54f9aa1a22c79adbb5c41535c6dd6950b9599097ac32222219256ba51ac6d0","observation_id":"2596427f-2627-40a1-8ddc-01f464ec602d","resolution":{"observed_at":"2026-08-11T22:51:50.610616Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:49.906849Z","title":"Natural TTS synthesis by conditioning Wavenet on mel-spectrogram predictions,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.906849Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:38fe4a6f39ba34e57bd9233ca230b6ada4b96da6470060f38b0422cd4dcd1dd9","observation_id":"6b041915-aab1-493c-951d-0ce34066dfd5","resolution":{"observed_at":"2026-08-11T22:51:49.906849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:49.912368Z","title":"On generative spoken language modeling from raw audio,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.912368Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:f131f66d28aa4241f96f9fc186b93f18c1796b1a8675eefdb645bb72183a92fe","observation_id":"4914c130-3b3c-4289-a90b-8357c6324bc6","resolution":{"observed_at":"2026-08-11T22:51:49.912368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1807.03748","last_updated":"2019-01-22T18:47:12Z","snapshot_observed_at":"2026-08-14T18:53:38.574749Z","submitted_at":"2018-07-10T16:52:11Z","title":"Representation Learning with Contrastive Predictive Coding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1807.03748","snapshot_observed_at":"2026-08-11T22:51:49.918052Z","title":"Representa- tion learning with contrastive predictive coding,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.918052Z"},"links":{"cited_paper":"/paper/1807.03748","citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:a7a267857673b3cc9fe8efdfff2d24f5e3e4871d2c19c721c1d6984ecc49d75c","observation_id":"c9f2b912-7f0a-48b6-8c94-7df3355f3f90","resolution":{"observed_at":"2026-08-11T22:51:49.918052Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:49.926589Z","title":"Wav2vec 2.0: A framework for self-supervised learning of speech representations,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.926589Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:5c4f98e194c8658c12bcc1c1389592af884809257d6ed2da4c6fb1bbf0951fad","observation_id":"7fd38301-1d33-4570-8c13-dad36764ceb5","resolution":{"observed_at":"2026-08-11T22:51:49.926589Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:49.932289Z","title":"HuBERT: Self- supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.932289Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:2645f9bcd9efefd6c688ba941bfd2249a3147f5c841adcdb530c2f7e973238cb","observation_id":"a807a468-54ee-468e-a89b-502e821a3a4e","resolution":{"observed_at":"2026-08-11T22:51:49.932289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.525657Z","title":"Attention is all you need,","venue":null,"work_id":"5651d731-fc51-4406-8cbd-bd87c4e71ea0","year":2017},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.937783Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:5edb17e1169a1df4453e2ab070e7cc12af2cd5721a8851eda875721144dc1de4","observation_id":"424690e8-0ec4-49c8-bbfc-5cd567793e8c","resolution":{"observed_at":"2026-08-11T22:51:50.532944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.507460Z","title":"The Zero Resource Speech Challenge 2020: Discovering Discrete Subword and Word Units,","venue":null,"work_id":"5d87bc1b-3777-43cd-9e64-07814c78e72b","year":2020},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.943099Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:717b2cb5085cec41b5e3e5634aee735376445c1d716db2f92cfd8acc42d120c4","observation_id":"62613339-6e88-422d-b6f7-bb67e2a1afbc","resolution":{"observed_at":"2026-08-11T22:51:50.513108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.488706Z","title":"The zero resource speech challenge 2021: Spoken language mod- elling,","venue":null,"work_id":"29817c98-8692-4d9e-913f-6546a5df0a3a","year":2021},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.948890Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:c777f1ee4803cad5b608f1f689448a50632bb46be88e0b97efe2f3b96a8d803d","observation_id":"426bac3e-2054-43a0-9b94-25b6876fb6bf","resolution":{"observed_at":"2026-08-11T22:51:50.494312Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.470685Z","title":"Self- supervised language learning from raw audio: Lessons from the zero resource speech challenge,","venue":null,"work_id":"91fa2cf8-d355-43eb-9fec-00b3e1594037","year":2022},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.954166Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:24e24b1ad764aebba732edaf55b0289009ff9f71b3dadf33cd0783dc101c0958","observation_id":"9e2a6a93-268c-4b36-aaeb-7d701ee17560","resolution":{"observed_at":"2026-08-11T22:51:50.476981Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.451073Z","title":"A comparison of discrete and soft speech units for improved voice conversion,","venue":null,"work_id":"63936888-ed4c-4576-a53c-286e2b836f73","year":2022},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.958904Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:1c2fb64cae1f54e1e57794414ea38b875c787b684d5a44238472a7ab6bdd17b1","observation_id":"afa51aa1-87a9-461f-9db8-b84826267a1a","resolution":{"observed_at":"2026-08-11T22:51:50.457515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.431090Z","title":"Textless direct speech-to- speech translation with discrete speech representation,","venue":null,"work_id":"08b23f21-bacc-4858-8b16-843ccbc8f8ca","year":2023},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.963908Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:37db0b77eb4e4a7ddfa412486e5db2680b411faea31d20db0b69a3fd64baaa2c","observation_id":"b10deed6-9dcb-4f28-95ce-5f99ed039354","resolution":{"observed_at":"2026-08-11T22:51:50.436813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.407532Z","title":"UnitSpeech: Speaker-adaptive Speech Synthesis with Untranscribed Data,","venue":null,"work_id":"46922252-b465-4c54-8e81-9c76e23d5515","year":2023},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.969832Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:5ae55208717e6c66692a558db868c844bf22c824f559be37c7f4a72ddb4e6752","observation_id":"af9ee82f-fc6c-41e9-8eba-ca1a7d11538d","resolution":{"observed_at":"2026-08-11T22:51:50.414008Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.389072Z","title":"LibriSpeech: An ASR corpus based on public domain audio books,","venue":null,"work_id":"dcc5ec44-04f4-4cea-a79e-b5cbc9e9ccec","year":2015},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.982938Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:48df77d9bef95c2e232f220187ddcc914c40678195f7c4fa1d416a16472015f5","observation_id":"be39764a-d380-48bb-bc88-d67353c8bfb5","resolution":{"observed_at":"2026-08-11T22:51:50.394880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.367242Z","title":"Ito and L","venue":null,"work_id":"4c8c8deb-174a-48d0-a4ac-6f69e24d93cf","year":2017},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.987883Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:5a7cde60c9898b644b8f4854f7be3dfb6f7944625a1602042babe7db4172b9d4","observation_id":"3b8d78e1-02aa-4d7d-9084-518504527449","resolution":{"observed_at":"2026-08-11T22:51:50.373300Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.348761Z","title":null,"venue":null,"work_id":"532fd4b8-301d-4fa8-8d59-53f44ac302df","year":2023},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.992872Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:fc7fed936e16afc30926340acd59533e438d4458ccf81aea6c854ed1658f13e7","observation_id":"26165c73-cf25-42f4-95ea-76ade09ae3eb","resolution":{"observed_at":"2026-08-11T22:51:50.354281Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.00354","last_updated":"2017-10-28T05:28:01Z","snapshot_observed_at":"2026-08-14T20:19:26.882819Z","submitted_at":"2017-10-28T05:28:01Z","title":"JSUT corpus: free large-scale Japanese speech corpus for end-to-end speech synthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.00354","snapshot_observed_at":"2026-08-11T22:51:49.998068Z","title":"JSUT corpus: Free large-scale Japanese speech corpus for end- to-end speech synthesis,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.998068Z"},"links":{"cited_paper":"/paper/1711.00354","citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:59473a40880617b26b3803b6c537e1e3d01d759f1032f45b0c5090c232c9de5c","observation_id":"9af1f3d6-57ea-4a1c-979a-6a410715f544","resolution":{"observed_at":"2026-08-11T22:51:49.998068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.06248","last_updated":"2019-08-17T06:04:46Z","snapshot_observed_at":"2026-08-14T12:49:46.383600Z","submitted_at":"2019-08-17T06:04:46Z","title":"JVS corpus: free Japanese multi-speaker voice corpus","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.06248","snapshot_observed_at":"2026-08-11T22:51:50.005341Z","title":"JVS corpus: Free Japanese multi- speaker voice corpus,","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:50.005341Z"},"links":{"cited_paper":"/paper/1908.06248","citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:e5cbb504b152d277283c5703bc586129cae2a179a7f67024bec7fd2884b8ed52","observation_id":"aba34485-f38b-4812-a5ff-9bf6ab45dbce","resolution":{"observed_at":"2026-08-11T22:51:50.005341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.011574Z","title":"Audio- book speech synthesis conditioned by cross-sentence context-aware word embeddings,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:50.011574Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:b82f5dae78acbfb0e094e94991590b50ac17822bf635169813ba3b8348f403e1","observation_id":"084663e5-cfc0-46c9-b2b8-d3d1fff071c9","resolution":{"observed_at":"2026-08-11T22:51:50.011574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.317616Z","title":"J-MAC: Japanese multi-speaker audiobook corpus for speech synthesis,","venue":null,"work_id":"9ae2d030-962e-4398-81ad-51c91be9b685","year":2022},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:50.016467Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:5a9d001238106a5266d1ad202eea68f0e5aa8c76d4732795dc5f4fa529b7bb3f","observation_id":"904682ab-e720-4882-9125-c6a52e1496ea","resolution":{"observed_at":"2026-08-11T22:51:50.323938Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.022731Z","title":"Radford, K","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:50.022731Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:f507c71ac231e6a88890b5bc7e366b75219900f35139bb6a90c7eaed9a8d9419","observation_id":"27ce4de1-9389-43c3-ae3a-80f37022a628","resolution":{"observed_at":"2026-08-11T22:51:50.022731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.289034Z","title":"Investi- gation of robustness of hubert features from different layers to domain, accent and language variations,","venue":null,"work_id":"1b766b10-4dd6-4ffe-8ccd-6b2866bdda10","year":2022},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:50.027954Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:934ca77e0cf9d049adb1777880cc53d944ea88abfb685252e6098d1f3d5edb23","observation_id":"7b2727b2-2a2b-486f-a526-48cf884fe8a8","resolution":{"observed_at":"2026-08-11T22:51:50.294196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.271310Z","title":"ContentVec: An improved self-supervised speech representation by dis- entangling speakers,","venue":null,"work_id":"16b2fca0-b648-4484-a1a0-e27e75234acd","year":2022},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:50.033061Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:42fa84bdd9dde3d82119b53132049662069d91c6b80aee38a2b6f74fa70122de","observation_id":"af4827dd-6f50-4f16-99e2-0a9732380725","resolution":{"observed_at":"2026-08-11T22:51:50.277654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.01793","last_updated":"2020-10-05T05:46:10Z","snapshot_observed_at":"2026-08-12T08:42:33.139142Z","submitted_at":"2020-10-05T05:46:10Z","title":"JSSS: free Japanese speech corpus for summarization and simplification","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.01793","snapshot_observed_at":"2026-08-11T22:51:50.039933Z","title":"JSSS: free japanese speech corpus for summarization and simplification,","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:50.039933Z"},"links":{"cited_paper":"/paper/2010.01793","citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:d855b005e529a0c26a3745b298282270b2d7d27f0cb341723c649411e07a3bc2","observation_id":"c61e9b87-1dbc-4368-81d9-e557c5c21a8c","resolution":{"observed_at":"2026-08-11T22:51:50.039933Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.253548Z","title":"Universal phone recognition with a multilingual allophone system,","venue":null,"work_id":"84a20443-54ae-4409-a2cb-823ea7e20600","year":2020},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:50.045137Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:58b6c03f610ef7d9027dc585445707411b38b160319f37a621067dbce13c1ca7","observation_id":"3828493d-eb62-45b8-a824-91f186f82db4","resolution":{"observed_at":"2026-08-11T22:51:50.258903Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.234357Z","title":"Speech quality assessment with W ARP-Q: From sim- ilarity to subsequence dynamic time warp cost,","venue":null,"work_id":"3e86a90a-3a1d-436f-b54f-d29bf72f7b06","year":2022},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:50.050472Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:c13878937b4a95a3999d8764ad9020ce3efa187dae18894ececa7f873dc4298b","observation_id":"7aa3af17-3970-46e3-8975-52bbd7399cf4","resolution":{"observed_at":"2026-08-11T22:51:50.240815Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21437/interspeech.2023-2326","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T22:51:50.084269Z","title":null,"venue":null,"work_id":"d3b64818-cf2e-4b75-ad9d-45f1121f0330","year":2023},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":3042,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.977002Z"},"links":{"citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:4a7c501c367e76a8a04089a53e21888f3e9e6cff97f63f62d932c6e282155275","observation_id":"a362ed75-5413-4e62-9c3f-f8c4e3014f7f","resolution":{"observed_at":"2026-08-11T22:51:50.091253Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-14T04:32:16.890178Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model"},"reference_resolution":{"displayed":30,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":14,"verified_exact":1,"verified_fuzzy":15},"total_outbound_references":30},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 30 of 30 outbound references and 0 inbound Pith citation observations for arXiv:2412.03074."}