{"as_of":"2026-08-19T06:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f7b83cfc0b18e87d19487635e9aa1ba9a1020a2209bb27411acc6315277dd1f8","coverage":[{"denominator":27,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T12:40:10.557940Z","state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2509.01391/citation-record","integrity":"/paper/2509.01391/integrity","json":"/paper/2509.01391/citation-record.json","paper":"/paper/2509.01391"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2203.01829","last_updated":"2022-03-01T11:15:35Z","snapshot_observed_at":"2026-08-16T17:17:51.073955Z","submitted_at":"2022-03-01T11:15:35Z","title":"A Brief Overview of Unsupervised Neural Speech Representation Learning","version":1},"cited_work":{"arxiv_id":"2203.01829","doi":null,"metadata_source":"pith","pith_arxiv_id":"2203.01829","snapshot_observed_at":"2026-08-05T12:40:10.673680Z","title":"A Brief Overview of Unsupervised Neural Speech Representation Learning","venue":"eess.AS","work_id":"6291de58-d3c7-4675-9c11-1ce25e18ce64","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.427945Z"},"links":{"cited_paper":"/paper/2203.01829","citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:bba9bc4f8681b97e23d363ede18e429e67a97aa4de5a55bec95eb5bb090deee9","observation_id":"5ff54f5e-682b-4d85-bd31-17cf031384a9","resolution":{"observed_at":"2026-08-05T12:40:10.681418Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:11.061493Z","title":"Natural TTS synthesis by conditioning Wavenet on mel-spectrogram predictions,","venue":null,"work_id":"a7a02e4e-1456-411b-b661-9f970627cad6","year":2018},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.433506Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:2825dce368b164c5f2dd4e8dc57596c2e12bd8b4e8353778d328ba98cef74a73","observation_id":"90f5f69d-626c-4cd5-ae28-15d4a175a960","resolution":{"observed_at":"2026-08-05T12:40:11.066460Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:11.044007Z","title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech,","venue":null,"work_id":"3e22516d-71e2-406b-b20d-198c4d63a1a5","year":2021},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.439036Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:d987d9d3d5d743f1cdaae8fff23f5cfdf345e3d2949ddb3d11bb489a4ce92621","observation_id":"8e44e2db-0d4c-446a-8314-cd30ad8b20d0","resolution":{"observed_at":"2026-08-05T12:40:11.049538Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:11.025029Z","title":"On generative spoken language modeling from raw audio,","venue":null,"work_id":"899d6ac9-1491-48c1-b0d6-2a9a541d94a3","year":2021},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.444464Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:d30b556feb3de1048dcfcdd3bc20c1179ed91aa620bb8ed24024a0a1f376dc7e","observation_id":"9172d4dd-ad1a-421f-b1c9-3208c80ac016","resolution":{"observed_at":"2026-08-05T12:40:11.031437Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:11.008124Z","title":"Wav2vec 2.0: A framework for self-supervised learning of speech representations,","venue":null,"work_id":"bef77f70-2e96-40c2-9a9c-2f053f4a0c19","year":2020},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.450324Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:cf5c8dc5a35ea2d086df561ad3fab331aff146fc2f0a000f4f572d255975c016","observation_id":"e34ba3e0-1027-4ac8-8070-ab1221fa4917","resolution":{"observed_at":"2026-08-05T12:40:11.013639Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.991441Z","title":"HuBERT: Self- supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":"c8e78147-5e72-476a-88ed-80cc39364a29","year":2021},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.455036Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:7806396ae2bae9892d7f297537d55bab140873a86a05db5d3347253c9b25b112","observation_id":"c6abd5db-d1d3-4674-9079-3c8f01a02db8","resolution":{"observed_at":"2026-08-05T12:40:10.996936Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.974180Z","title":"A Unified Accent Estimation Method Based on Multi-Task Learn- ing for Japanese Text-to-Speech,","venue":null,"work_id":"26663653-70db-4d10-8bce-bbea5f72ea90","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.460355Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:aa6129010244cbbd20556bc5427090fc0ea939882d4d8f950b4a21db650bacd9","observation_id":"5e76a905-5d1d-4d2e-a1bb-11c2f2f79f8e","resolution":{"observed_at":"2026-08-05T12:40:10.979543Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.956871Z","title":"Reazonspeech: A free and massive corpus for Japanese ASR,","venue":null,"work_id":"9473f678-e372-4b89-bbb3-fb7e9196a88c","year":2023},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.465377Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:fd4be1e44eb96d05627bdb23e2f537c65b780699eca8e380a380fab84320ee61","observation_id":"c4c91880-fe62-47fb-9d30-c05813665721","resolution":{"observed_at":"2026-08-05T12:40:10.961640Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.938040Z","title":"Exploring the limits of transfer learning with a unified text-to- text transformer,","venue":null,"work_id":"890306a8-c90c-4458-8888-1d04c5fa09ae","year":2019},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.469772Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:74c20d45fb102b77bedb1b3004e40940d9a811bc8cbeada09ccc5f42bc67bcd7","observation_id":"daf8e8ad-0319-407d-99de-c4f775461830","resolution":{"observed_at":"2026-08-05T12:40:10.943901Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.00354","last_updated":"2017-10-28T05:28:01Z","snapshot_observed_at":"2026-08-14T20:19:26.882819Z","submitted_at":"2017-10-28T05:28:01Z","title":"JSUT corpus: free large-scale Japanese speech corpus for end-to-end speech synthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.00354","snapshot_observed_at":"2026-08-05T12:40:10.474344Z","title":"JSUT corpus: Free large-scale Japanese speech corpus for end- to-end speech synthesis,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.474344Z"},"links":{"cited_paper":"/paper/1711.00354","citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:d3964f7e7e2784c637e05fea67337961ef638eb3e849b1a8e7f510b500c46ed4","observation_id":"4326f94e-a02e-415e-a5c6-8d8e5a9fc790","resolution":{"observed_at":"2026-08-05T12:40:10.474344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.06248","last_updated":"2019-08-17T06:04:46Z","snapshot_observed_at":"2026-08-17T09:27:25.572276Z","submitted_at":"2019-08-17T06:04:46Z","title":"JVS corpus: free Japanese multi-speaker voice corpus","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.06248","snapshot_observed_at":"2026-08-05T12:40:10.479287Z","title":"JVS corpus: Free Japanese multi- speaker voice corpus,","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.479287Z"},"links":{"cited_paper":"/paper/1908.06248","citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:4bf010779486da479110d68144e2b0b4bc0bfd548be4bd7d99a38396521dbedd","observation_id":"7237d5cf-fe98-43df-a014-99330b8e6193","resolution":{"observed_at":"2026-08-05T12:40:10.479287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.921378Z","title":"Audio- book speech synthesis conditioned by cross-sentence context-aware word embeddings,","venue":null,"work_id":"f4099217-74f7-469d-bb78-5c84ce72c616","year":2021},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.484916Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:d6931d320b34e5942fdb9d7cb296855c0e6ee03ae5dd325e83528e7044e6d612","observation_id":"02a76c6d-7964-40a0-8132-93a3e365dd55","resolution":{"observed_at":"2026-08-05T12:40:10.926464Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.903218Z","title":"J-MAC: Japanese multi-speaker audiobook corpus for speech synthesis,","venue":null,"work_id":"f8f642f9-3c96-4041-92bb-f4a4b0855c70","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.489843Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:4643fe70b7fb088e291df1ebe11320aecfda433f73c07fbd3d90abe0a4d711e6","observation_id":"9d4ba3ca-6227-4aa6-8e29-535cc43dc83b","resolution":{"observed_at":"2026-08-05T12:40:10.908829Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.01793","last_updated":"2020-10-05T05:46:10Z","snapshot_observed_at":"2026-08-16T19:15:55.699345Z","submitted_at":"2020-10-05T05:46:10Z","title":"JSSS: free Japanese speech corpus for summarization and simplification","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.01793","snapshot_observed_at":"2026-08-05T12:40:10.494546Z","title":"JSSS: free japanese speech corpus for summarization and simplification,","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.494546Z"},"links":{"cited_paper":"/paper/2010.01793","citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:0bbe22c8d80e5eae5c105d8b3e5272b8a7eb064dc5e0148c1aafc831e8e618d9","observation_id":"f41b6dad-0f1d-4279-8df4-09aa58a892d9","resolution":{"observed_at":"2026-08-05T12:40:10.494546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.885297Z","title":"T5g2p: Us- ing text-to-text transfer transformer for grapheme-to- phoneme conversion,","venue":null,"work_id":"0dd1280b-f10c-489d-b272-eda3ce566818","year":2021},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.500200Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:63d4d77988fa4535f032a82fdb6dd948df4f29c89feec9b0183e8fc41faa8fe5","observation_id":"0d808ecc-efea-487b-9f18-88eb319e06ef","resolution":{"observed_at":"2026-08-05T12:40:10.890489Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.867472Z","title":"SpeechT5: Unified- modal encoder-decoder pre-training for spoken language processing,","venue":null,"work_id":"82fe7e8a-e655-4ff7-94cc-1437cdd4fe05","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.504454Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:8d20a76abc08538789aab822a3567b2749b86d88a81a8b3e52e51d50db79416a","observation_id":"5af588d6-e2ec-44fe-819d-ace2c0e3a93e","resolution":{"observed_at":"2026-08-05T12:40:10.872317Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.851394Z","title":"Neural ma- chine translation for multilingual grapheme-to-phoneme conversion,","venue":null,"work_id":"5e7bdbf8-16de-4ca2-834e-cbb9d6bdf11c","year":2019},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.510410Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:76e6852e7e0f16284d6f03e7dac086086fc9453b2cc3ae54bb65603317c33a8c","observation_id":"36323435-2a5d-4930-81db-cf68a5d70b52","resolution":{"observed_at":"2026-08-05T12:40:10.856534Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.833643Z","title":"One model to pronounce them all: Multilingual grapheme- to-phoneme conversion with a transformer ensemble,","venue":null,"work_id":"6ecb89f6-0d67-41d3-8b25-7a6ffff0e0d2","year":2020},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.515754Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:5c6628d6fbb9bb3f7396be05669fbfa8e96e77cbb7b440855cadf1aac2456882","observation_id":"13135d83-6c53-4f40-81dc-c563038ebeb5","resolution":{"observed_at":"2026-08-05T12:40:10.838454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.815490Z","title":"Byt5 model for mas- sively multilingual grapheme-to-phoneme conversion,","venue":null,"work_id":"6180b7fb-b1d3-4d5c-8bd1-66707f640e00","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.520427Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:6178e9de70b96e39d80266c622a5084b2f4de62788df0d74578c8abd35e5f23c","observation_id":"d50b1644-9ac2-43ee-bb72-12798424b365","resolution":{"observed_at":"2026-08-05T12:40:10.820105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.799243Z","title":"ContentVec: An improved self-supervised speech representation by dis- entangling speakers,","venue":null,"work_id":"740361a7-3577-402a-8f5b-df726a6becc4","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.524640Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:90f028521175759e4ae064db76cead7a158379dfa3cc2cf71bef8cccc26cc99d","observation_id":"369ce2af-1ef8-40ed-84bf-32853dd86534","resolution":{"observed_at":"2026-08-05T12:40:10.803937Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-14T02:11:04.009144Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-05T12:40:10.529671Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.529671Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:c45a7b4d9d443e963eaa08c70fda87b87858699c7bbb57bbb0fc24d65d9f058a","observation_id":"1257031f-77b7-42fc-a4b6-f15c4431988c","resolution":{"observed_at":"2026-08-05T12:40:10.529671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.780708Z","title":"Radford, K","venue":null,"work_id":"db1c4654-dc68-45c1-a362-483aa4bab2b3","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.534763Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:64120c5815f32a15bbe459cb3f069b0436d0e7f02d5692b0eb6db68bb7674d3e","observation_id":"b7c8b8da-ac01-4958-b309-6a285538d60a","resolution":{"observed_at":"2026-08-05T12:40:10.785436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.763088Z","title":"UTMOS: UTokyo-SaruLab system for VoiceMOS challenge 2022,","venue":null,"work_id":"dcad5633-b255-4eee-bb13-8ad7d8b6226d","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.539047Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:ef62ca37836321bea29644783e382dfcded77fe79da9a2fe4d62f20e900c2fd1","observation_id":"e515bfbf-108a-4b89-929a-49b24a59a620","resolution":{"observed_at":"2026-08-05T12:40:10.768827Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.745533Z","title":"Speech quality assessment with W ARP-Q: From sim- ilarity to subsequence dynamic time warp cost,","venue":null,"work_id":"060e3f32-b69d-409a-867b-b4948bf90bfe","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.543959Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:ae58026335da83da937466b63606366ca2523f836d5631bc7dc24dfbdfe38413","observation_id":"2ae9a126-9e21-46af-a0d5-77bf7ccb7e17","resolution":{"observed_at":"2026-08-05T12:40:10.750702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.726743Z","title":"Per- ceptual evaluation of speech quality (PESQ)-a new method for speech quality assessment of telephone networks and codecs,","venue":null,"work_id":"7d241002-8a56-4a2a-bb8b-fd698a476cd3","year":2001},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.548721Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:74e33d3a58ea583b846080b55802cc4cded67d928f9df26fd74dd2d7de14685a","observation_id":"90e04134-54e2-4c17-8272-80191bbcc018","resolution":{"observed_at":"2026-08-05T12:40:10.732694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.708663Z","title":"mT5: A mas- sively multilingual pre-trained text-to-text transformer,","venue":null,"work_id":"8be60b7b-b938-4452-9d33-c59929d6a86c","year":2021},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.553662Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:a2dcd2af9ffa94401a2e712e876ce98fc10986e9d6506a8b39ef364e46bc836d","observation_id":"07250b96-e2a1-4327-9ce1-90001fc03b31","resolution":{"observed_at":"2026-08-05T12:40:10.713327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T12:40:10.693256Z","title":"ByT5: Towards a token-free future with pre-trained byte-to-byte models,","venue":null,"work_id":"aa553116-8af1-4b5e-99da-778a87de4f1a","year":2022},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.557940Z"},"links":{"citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:13bb7a2fd09dea662ddc8d42783f7d3f1bc93cb9840606810d8f02d62188f983","observation_id":"d9cf1ded-3c45-4a9f-957d-e7bd01a8cb69","resolution":{"observed_at":"2026-08-05T12:40:10.697841Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model"},"reference_resolution":{"displayed":27,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":4,"verified_exact":1,"verified_fuzzy":22},"total_outbound_references":27},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 27 of 27 outbound references and 0 inbound Pith citation observations for arXiv:2509.01391."}