{"as_of":"2026-08-20T04:49:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7853304fd1be6c690c65d674ef381fd7175d25b8c79d32ffef5bc1b23a7149f0","coverage":[{"denominator":40,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":40,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:02:44.434638Z","state":"measured"},{"denominator":48,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":48,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":8,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":8,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T08:34:04.507414Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T07:56:57.825825Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"cited_work":{"arxiv_id":"2506.11130","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.11130","snapshot_observed_at":"2026-07-10T07:56:57.825825Z","title":"A self-refining frame- work for enhancing asr using tts-synthesized data","venue":"cs.CL","work_id":"73a675bd-873c-4c68-ae94-7772f8941f1b","year":2025},"citing_paper":{"arxiv_id":"2604.10065","last_updated":"2026-04-11T07:07:08Z","snapshot_observed_at":"2026-08-13T18:44:53.184030Z","submitted_at":"2026-04-11T07:07:08Z","title":"ASPIRin: Action Space Projection for Interactivity-Optimized Reinforcement Learning in Full-Duplex Speech Language Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T17:05:45.214298Z"},"links":{"cited_paper":"/paper/2506.11130","citing_paper":"/paper/2604.10065"},"observation_digest":"sha256:edbb55753c05607ee86796c527de3d76e88f7a33e4a0266b34d90e01c65fcd77","observation_id":"66bc31a7-5e68-4c2b-ad52-848df8d5018e","resolution":{"observed_at":"2026-05-11T07:35:58.880514Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"cited_work":{"arxiv_id":"2506.11130","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.11130","snapshot_observed_at":"2026-07-10T07:56:57.825825Z","title":"A self-refining frame- work for enhancing asr using tts-synthesized data","venue":"cs.CL","work_id":"73a675bd-873c-4c68-ae94-7772f8941f1b","year":2025},"citing_paper":{"arxiv_id":"2606.29031","last_updated":"2026-07-09T12:32:31Z","snapshot_observed_at":"2026-08-15T14:29:36.597543Z","submitted_at":"2026-06-27T17:57:27Z","title":"How to Leverage Synthetic Speech for LLM-Based ASR Systems?","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-30T09:35:03.660671Z"},"links":{"cited_paper":"/paper/2506.11130","citing_paper":"/paper/2606.29031"},"observation_digest":"sha256:fd559e66a7ace2ccd7f4b67565ed1e8de22c929b734f370394003f9db89411f2","observation_id":"7b470afe-9f7a-4760-9ff1-a46b8f602637","resolution":{"observed_at":"2026-06-30T12:54:40.707154Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.11130","snapshot_observed_at":"2026-07-12T11:11:08.878850Z","title":"A self-refining framework for enhancing asr using tts-synthesized data,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.29031","last_updated":"2026-07-09T12:32:31Z","snapshot_observed_at":"2026-08-15T14:29:36.597543Z","submitted_at":"2026-06-27T17:57:27Z","title":"How to Leverage Synthetic Speech for LLM-Based ASR Systems?","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-12T11:11:08.878850Z"},"links":{"cited_paper":"/paper/2506.11130","citing_paper":"/paper/2606.29031"},"observation_digest":"sha256:8aabb7f93c372396e623fb06fefb788d2f6bd5a4e319bae96eab12e53270fd3f","observation_id":"3d6112c8-506e-4003-9006-343e34a6122f","resolution":{"observed_at":"2026-07-12T11:11:08.878850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.11130","snapshot_observed_at":"2026-07-11T09:26:54.286790Z","title":"A self-refining framework for enhancing ASR using TTS-synthesized data,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.05058","last_updated":"2026-07-06T13:31:09Z","snapshot_observed_at":"2026-08-15T09:41:22.709382Z","submitted_at":"2026-07-06T13:31:09Z","title":"Context-Aware ASR for Mandarin Technical Lectures","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-11T09:26:54.286790Z"},"links":{"cited_paper":"/paper/2506.11130","citing_paper":"/paper/2607.05058"},"observation_digest":"sha256:5e3f55dcd6a0b5d2e8ef86e3e922b20f6e77174b2009f78fb14cdbaa77465de8","observation_id":"44fc9157-acf4-47d7-b5b9-92517e912a3e","resolution":{"observed_at":"2026-07-11T09:26:54.286790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"cited_work":{"arxiv_id":"2506.11130","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.11130","snapshot_observed_at":"2026-07-10T07:56:57.825825Z","title":"A self-refining frame- work for enhancing asr using tts-synthesized data","venue":"cs.CL","work_id":"73a675bd-873c-4c68-ae94-7772f8941f1b","year":2025},"citing_paper":{"arxiv_id":"2607.05364","last_updated":"2026-07-15T17:16:51Z","snapshot_observed_at":"2026-08-15T21:16:03.653789Z","submitted_at":"2026-07-06T17:40:54Z","title":"REDDIT: Correcting Model-Generated Timestamp Drift in ASR without Forgetting via Replay-Based Distribution Editing","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-07T14:52:35.127533Z"},"links":{"cited_paper":"/paper/2506.11130","citing_paper":"/paper/2607.05364"},"observation_digest":"sha256:d2f17a02780055738541adea5a389650950da2d0bb6322e43bde1a048d25e060","observation_id":"ac312fe2-935e-4db9-a082-1052100c4ef5","resolution":{"observed_at":"2026-07-07T14:53:55.858603Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.11130","snapshot_observed_at":"2026-08-02T08:34:04.507414Z","title":"A self-refining framework for enhanc- ing ASR using TTS-synthesized data,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.05364","last_updated":"2026-07-15T17:16:51Z","snapshot_observed_at":"2026-08-15T21:16:03.653789Z","submitted_at":"2026-07-06T17:40:54Z","title":"REDDIT: Correcting Model-Generated Timestamp Drift in ASR without Forgetting via Replay-Based Distribution Editing","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-02T08:34:04.507414Z"},"links":{"cited_paper":"/paper/2506.11130","citing_paper":"/paper/2607.05364"},"observation_digest":"sha256:25e66b1ff1f9e7c72853e782861b4f92179a1012ed13f018ad9aebe652d72730","observation_id":"219c8161-cbe9-43e2-a9b9-94e4ba10c391","resolution":{"observed_at":"2026-08-02T08:34:04.507414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"cited_work":{"arxiv_id":"2506.11130","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.11130","snapshot_observed_at":"2026-07-10T07:56:57.825825Z","title":"A self-refining frame- work for enhancing asr using tts-synthesized data","venue":"cs.CL","work_id":"73a675bd-873c-4c68-ae94-7772f8941f1b","year":2025},"citing_paper":{"arxiv_id":"2607.06054","last_updated":"2026-07-07T09:31:19Z","snapshot_observed_at":"2026-08-14T10:10:54.178445Z","submitted_at":"2026-07-07T09:31:19Z","title":"BlueMagpie-TTS: A Token-Efficient Tokenizer, Language Model, and TTS for Taiwanese-Accent Code-Switching Speech","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-08T18:09:22.207379Z"},"links":{"cited_paper":"/paper/2506.11130","citing_paper":"/paper/2607.06054"},"observation_digest":"sha256:5f296c6c45a1bc91e7846122e11a56c50b12483c0f7ce0994d23cdecbf2e8411","observation_id":"40bfac51-d6fb-42a3-9f54-3faa9e183317","resolution":{"observed_at":"2026-07-08T18:15:21.432691Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"cited_work":{"arxiv_id":"2506.11130","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.11130","snapshot_observed_at":"2026-07-10T07:56:57.825825Z","title":"A self-refining frame- work for enhancing asr using tts-synthesized data","venue":"cs.CL","work_id":"73a675bd-873c-4c68-ae94-7772f8941f1b","year":2025},"citing_paper":{"arxiv_id":"2607.08409","last_updated":"2026-07-09T12:34:56Z","snapshot_observed_at":"2026-08-08T18:32:53.947470Z","submitted_at":"2026-07-09T12:34:56Z","title":"When Synthetic Speech Is All You Have: Better Call GRPO","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-10T07:53:41.580597Z"},"links":{"cited_paper":"/paper/2506.11130","citing_paper":"/paper/2607.08409"},"observation_digest":"sha256:23cd4774fbc11a1eef000035db14feba1c538fd2fc61aae7d431746ef96fab3b","observation_id":"f4c1c24d-f962-49e2-a2e3-79a370005f00","resolution":{"observed_at":"2026-07-10T07:56:57.826984Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.11130/citation-record","integrity":"/paper/2506.11130/integrity","json":"/paper/2506.11130/citation-record.json","paper":"/paper/2506.11130"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.857944Z","title":"Canary-1b,","venue":null,"work_id":"2dd9cc73-6941-46ef-801a-318b694d1eb9","year":2023},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:42.793832Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:5380213bdd1da05803a3f02d658b90568d6e91ce9e08c66d9c0c7caedb8efad6","observation_id":"09713c24-144e-4d48-ac34-f892f444cc1a","resolution":{"observed_at":"2026-08-07T05:02:44.860662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.849126Z","title":"Robust speech recognition via large- scale weak supervision,","venue":null,"work_id":"4f80a23a-f6f2-43e1-966a-c344c07ee203","year":2022},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:42.848259Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:93b053afca60df577af4f062a420f521aba93fe6952f87060e7c10c45ed66593","observation_id":"662533db-9e42-442e-af6d-a128959cbbbe","resolution":{"observed_at":"2026-08-07T05:02:44.852209Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01743","last_updated":"2025-03-07T09:05:58Z","snapshot_observed_at":"2026-08-15T22:57:45.773661Z","submitted_at":"2025-03-03T17:05:52Z","title":"Phi-4-Mini Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01743","snapshot_observed_at":"2026-08-07T05:02:42.960494Z","title":"Phi-4-mini technical report: Compact yet pow- erful multimodal language and multimodal models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:42.960494Z"},"links":{"cited_paper":"/paper/2503.01743","citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:0e42f8b58c0307489b490c8e19561a2549cee0da1d5a72e7552bbf4d28e1dbd0","observation_id":"104cc75b-bc4f-4529-8055-cebd11a13c5d","resolution":{"observed_at":"2026-08-07T05:02:42.960494Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.01037","last_updated":"2023-09-25T01:20:23Z","snapshot_observed_at":"2026-08-16T15:51:28.593814Z","submitted_at":"2023-03-02T07:47:18Z","title":"Google USM: Scaling Automatic Speech Recognition Beyond 100 Languages","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.01037","snapshot_observed_at":"2026-08-07T05:02:43.054744Z","title":"Usm: Scaling auto- matic speech recognition beyond 100 languages,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:43.054744Z"},"links":{"cited_paper":"/paper/2303.01037","citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:591c0ebf5a3e89c2d5e03d5743c7601c6c01b8f53058aa831c7a03389b271243","observation_id":"15f02759-05b4-497e-9bda-c790d208db1e","resolution":{"observed_at":"2026-08-07T05:02:43.054744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-07T05:02:43.165097Z","title":"Llama: Open and efficient foundation language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:43.165097Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:70a6021577088efe3d6c97789d3157c63ba9cd9c72067e86a56d98e151c7ed5b","observation_id":"0f229948-93ad-4221-a704-877f2e96ec7b","resolution":{"observed_at":"2026-08-07T05:02:43.165097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.840398Z","title":"Olmo: Accelerating the science of language models,","venue":null,"work_id":"2c4159f8-4871-448e-9444-24be8d5436b3","year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:43.255563Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:b8c7acfb339fcbf5062c6e1be07784734cb29e66cc5d5ce893d657dd0e3f30ed","observation_id":"0bda8398-d3db-4345-a7c1-161a1d882905","resolution":{"observed_at":"2026-08-07T05:02:44.843500Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.831766Z","title":"Using synthetic audio to improve the recognition of out-of-vocabulary words in end-to-end asr systems,","venue":null,"work_id":"91244ee2-98e3-41de-96aa-93ae180003b3","year":2021},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:43.319915Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:fac60415e2b2b82681cb5580301cc103594eee714a572dd5e7bff723a82fe7f6","observation_id":"4e756a88-b01c-4fcf-83d7-f6ffcb8777cc","resolution":{"observed_at":"2026-08-07T05:02:44.834919Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.822574Z","title":"Text-only domain adaptation for end-to-end asr using integrated text-to-mel-spectrogram generator,","venue":null,"work_id":"b6a8afe8-e922-4049-a110-f7a429738179","year":2023},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:43.432373Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:34626f55871184a6209ff00084b80a63755ef81551d7bbf0e80399a14cb82908","observation_id":"c3209f4f-15b8-43d3-98ec-6b8185379d5c","resolution":{"observed_at":"2026-08-07T05:02:44.825752Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.813220Z","title":"Text is all you need: Personalizing asr models using controllable speech synthesis,","venue":null,"work_id":"010b0032-8d5e-4672-a293-0cd15473af97","year":2023},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:43.458564Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:192a29e72930f062482a5ae2fc5b1d0b18d061bbeb51b16733cc528d317626ad","observation_id":"919869c1-63a6-4b95-87a9-eb9ac0d2469f","resolution":{"observed_at":"2026-08-07T05:02:44.816617Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.804572Z","title":"Corpus synthesis for zero-shot asr domain adaptation using large language models,","venue":null,"work_id":"cf8f65c6-820b-4cba-865a-a5224a39ac19","year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:43.572469Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:17980294556d553b9db842d856b06811df88e1064cfc59cc40c5b2e204a2fa5e","observation_id":"638f1136-88cc-470e-bc51-b5a45698d866","resolution":{"observed_at":"2026-08-07T05:02:44.807632Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.795372Z","title":"Task arithmetic can mitigate synthetic-to-real gap in automatic speech recognition,","venue":null,"work_id":"f6304f0d-17b7-476e-9531-f3d5f57f5171","year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:43.695199Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:ab85c2ae1cd4574bb8d4e8929c37901a208028c9b9e00a11e5401e2a3cd9b406","observation_id":"87178acd-1693-4f7f-b19f-03122e216dfb","resolution":{"observed_at":"2026-08-07T05:02:44.798599Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.786398Z","title":"Enhancing low-resource asr through versatile tts: Bridging the data gap,","venue":null,"work_id":"65ce9dd1-852b-49c3-8ec2-a1df383c394c","year":2025},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:43.828322Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:7c7983331c57b6b472508117228d7bb617e57da79e7f179d86e681149d72e8ec","observation_id":"f653ef1d-cd9b-4cca-bdd8-d6849a7ca75f","resolution":{"observed_at":"2026-08-07T05:02:44.789636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.776935Z","title":"Coqui tts,","venue":null,"work_id":"9d5a3950-0fe1-494d-bb7b-f29f8924cf71","year":2021},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:43.984291Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:98dc277f040caffd0fb5b2b83173c294b07fff3f016e7bdd34a0fe8c0fb9d151","observation_id":"1f28d616-630e-49f7-be2b-d40a4e527f94","resolution":{"observed_at":"2026-08-07T05:02:44.780191Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.767254Z","title":"Matcha-tts: A fast tts architecture with conditional flow matching,","venue":null,"work_id":"548c56a1-c264-45d7-9e2c-f3711fba07f3","year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.138447Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:e62f711347ef8ad0cec4cf1dc17a373d5182e53f0eff1b5c97fd31ca2f9df511","observation_id":"ac740d26-9413-4b1f-b83f-82037dfb0eff","resolution":{"observed_at":"2026-08-07T05:02:44.770859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-08-10T17:49:50.848957Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-07T05:02:44.246789Z","title":"Cosyvoice: A scalable multilingual zero-shot text-to-speech synthesizer based on supervised semantic tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.246789Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:40a30cdbcc1d9d403747ff7ee3eb9b5cce85f98b0269aa41c732a7e9a6b0643f","observation_id":"32169c15-5d06-4017-9e15-760990f98dc8","resolution":{"observed_at":"2026-08-07T05:02:44.246789Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17790","last_updated":"2025-01-29T17:31:26Z","snapshot_observed_at":"2026-08-19T11:44:24.722195Z","submitted_at":"2025-01-29T17:31:26Z","title":"BreezyVoice: Adapting TTS for Taiwanese Mandarin with Enhanced Polyphone Disambiguation -- Challenges and Insights","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17790","snapshot_observed_at":"2026-08-07T05:02:44.355741Z","title":"Breezyvoice: Adapting tts for taiwanese mandarin with enhanced polyphone disambiguation–challenges and insights,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.355741Z"},"links":{"cited_paper":"/paper/2501.17790","citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:4751d93c1ac7b0de2a8d4785d68082964b3924036365d5a0351afaafd7d04b21","observation_id":"ba4d3c8e-de90-4309-8294-03b2b6fa800a","resolution":{"observed_at":"2026-08-07T05:02:44.355741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.757417Z","title":"Leave no knowledge behind during knowledge distillation: Towards practical and effective knowledge distillation for code-switching asr using realistic data,","venue":null,"work_id":"624a711e-f1df-45a3-aea9-4cae57888b80","year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.359603Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:5b77befd7c1d78753b972d0d442ec28aa646060f54bd70fd3535afb224095d22","observation_id":"d0e7526e-f689-4e97-81cc-bdf2b866b07b","resolution":{"observed_at":"2026-08-07T05:02:44.760966Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.18485","last_updated":"2025-03-28T13:26:11Z","snapshot_observed_at":"2026-08-16T12:47:25.751387Z","submitted_at":"2025-03-24T09:39:41Z","title":"Whispering in Amharic: Fine-tuning Whisper for Low-resource Language","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.18485","snapshot_observed_at":"2026-08-07T05:02:44.363082Z","title":"Whispering in amharic: Fine-tuning whisper for low-resource language,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.363082Z"},"links":{"cited_paper":"/paper/2503.18485","citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:2a138fa73040b559c477a8d7ff941c8e7e78797e4bbd1bdc0eff9adb3ecbedda","observation_id":"4e6f13ba-272b-4c1a-bdee-4f22176ceb35","resolution":{"observed_at":"2026-08-07T05:02:44.363082Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.747565Z","title":"Whispering in norwegian: Navigating orthographic and dialectic challenges,","venue":null,"work_id":"bd930841-3a93-4b50-a936-42fdb7f942a8","year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.366290Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:d63980f12c6c6fb863b41d083445419f1633a93170ecfe45ec72ddcac10d3f9e","observation_id":"6d857015-18fb-448d-b0ba-4585d63eb3f2","resolution":{"observed_at":"2026-08-07T05:02:44.750751Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17284","last_updated":"2025-02-24T16:11:13Z","snapshot_observed_at":"2026-08-16T12:55:41.322721Z","submitted_at":"2025-02-24T16:11:13Z","title":"Improving the Inclusivity of Dutch Speech Recognition by Fine-tuning Whisper on the JASMIN-CGN Corpus","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.17284","snapshot_observed_at":"2026-08-07T05:02:44.369784Z","title":"Improving the inclusivity of dutch speech recognition by fine-tuning whisper on the jasmin-cgn corpus,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.369784Z"},"links":{"cited_paper":"/paper/2502.17284","citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:b719a09e7afd87ce9c8939505701f152d662f5da2b762931dbeed5d245c2fbc7","observation_id":"980ba760-e2d5-4d5d-bad9-30617b419090","resolution":{"observed_at":"2026-08-07T05:02:44.369784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.12587","last_updated":"2024-11-19T15:55:56Z","snapshot_observed_at":"2026-08-18T16:37:12.680040Z","submitted_at":"2024-11-19T15:55:56Z","title":"Whisper Finetuning on Nepali Language","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.12587","snapshot_observed_at":"2026-08-07T05:02:44.373293Z","title":"Whisper finetuning on nepali language,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.373293Z"},"links":{"cited_paper":"/paper/2411.12587","citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:03143567a70b3a0aa9666cd6d31948e69c60d1f0cb6b6de4fb1b1ebbd2497af6","observation_id":"934b101f-821e-41f5-bafa-13a4274c834b","resolution":{"observed_at":"2026-08-07T05:02:44.373293Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10705","last_updated":"2024-12-14T06:32:16Z","snapshot_observed_at":"2026-08-16T21:04:24.342651Z","submitted_at":"2024-12-14T06:32:16Z","title":"Efficient Adaptation of Multilingual Models for Japanese ASR","version":1},"cited_work":{"arxiv_id":"2412.10705","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.10705","snapshot_observed_at":"2026-08-07T05:02:44.475411Z","title":"Efficient Adaptation of Multilingual Models for Japanese ASR","venue":"cs.CL","work_id":"7987c73a-9533-4886-8a5f-3392d716ef44","year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.376241Z"},"links":{"cited_paper":"/paper/2412.10705","citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:6d7d0c67939c15508f6a7dfff3f8d8865719d970904839d88cb6055246228c03","observation_id":"2e399dec-e9d1-410c-9f25-1cdf3804fc0b","resolution":{"observed_at":"2026-08-07T05:02:44.481540Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15726","last_updated":"2025-04-22T12:09:39Z","snapshot_observed_at":"2026-08-19T00:00:04.504964Z","submitted_at":"2024-12-20T09:49:02Z","title":"Fine-tuning Whisper on Low-Resource Languages for Real-World Applications","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15726","snapshot_observed_at":"2026-08-07T05:02:44.379377Z","title":"Fine-tuning whisper on low-resource languages for real-world applications,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.379377Z"},"links":{"cited_paper":"/paper/2412.15726","citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:72cc6cfb27d80beabc8c790469ad2c7d29357954dc8132f7c5db2e05f4528b92","observation_id":"7d06113f-401f-4f05-87f1-0ed9bc90cad3","resolution":{"observed_at":"2026-08-07T05:02:44.379377Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.738244Z","title":"The NTNU ASR system for Formosa speech recognition challenge 2023,","venue":null,"work_id":"c609186c-b6b4-41cf-a981-63ae66085834","year":2023},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.382508Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:9fa1c62f4e2ad0f1dfd120b5083975afb38ed48612c6a4d26f1120284a3facc0","observation_id":"7f81d8e4-e523-4ca3-9c4c-e8179e59ff0d","resolution":{"observed_at":"2026-08-07T05:02:44.741425Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.726593Z","title":"Zero resource code-switched speech benchmark using speech utterance pairs for multiple spoken languages,","venue":null,"work_id":"66f76ebe-689a-4eb4-9974-19ae3ebb292b","year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.385418Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:464d5b78550134a95d0e4b826a78749d38f1ad78f8bef8be4d501d62d125752c","observation_id":"c062ea13-214c-407f-9d7b-e43ccb36cdb3","resolution":{"observed_at":"2026-08-07T05:02:44.730942Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.715860Z","title":"IEEE, 2024, pp","venue":null,"work_id":"a6da1b7d-af4d-4d78-aed8-7e64d89665cf","year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.388771Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:e047c0b1b4edcb4ec05e1f2ce559e285c14ec8ee68d13aec042a763267b92760","observation_id":"4a444104-473b-4f62-93c5-834d5a2416d5","resolution":{"observed_at":"2026-08-07T05:02:44.719164Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.706255Z","title":"Zero-shot domain-sensitive speech recognition with prompt-conditioning fine-tuning,","venue":null,"work_id":"c5c83ee3-3ba6-448a-91fd-c444c89a0519","year":2023},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.392048Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:e1037b24d44e3a318ea015e270e258cc0899c410ada7559f9a2d731e9187efff","observation_id":"cca4f21b-4b3f-470d-89e1-d445227237ff","resolution":{"observed_at":"2026-08-07T05:02:44.709818Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.695654Z","title":"Prompting the hidden talent of web-scale speech models for zero-shot task generalization,","venue":null,"work_id":"0a61bd98-0d47-4131-8543-649da3766af6","year":2023},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.395736Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:a75f47494acbe1c7687e4722b287fc6dfde861954d25cfea6b33d5c0f9838a9e","observation_id":"72df09f8-47ad-44b5-ada4-b9e1644f1054","resolution":{"observed_at":"2026-08-07T05:02:44.700018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.685718Z","title":"Can whisper perform speech-based in-context learning?,","venue":null,"work_id":"298ea5ed-d1a4-4b8a-9577-4f344664f210","year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.399184Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:d059fdd41174078bac95f92c302354183cd52637cc1acf46e3130c8a8b138cdc","observation_id":"b80664d1-98f3-40ad-8896-11761ffe46ea","resolution":{"observed_at":"2026-08-07T05:02:44.689371Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.674787Z","title":"High fidelity neural audio compression,","venue":null,"work_id":"0cf5e226-633e-41ac-b2be-8fdf245b2533","year":null},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.402693Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:0dca7a306d4b02ad75749038dcaf876467ed2db6da2299934cadb874e6fd66b2","observation_id":"c4ad91b3-e85a-4ecb-afac-f6d26d9cd3db","resolution":{"observed_at":"2026-08-07T05:02:44.678618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.664316Z","title":"Funcodec: A fundamental, reproducible and integrable open-source toolkit for neural speech codec,","venue":null,"work_id":"7258bdf3-454d-400e-a69f-3b87901c7162","year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.406127Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:010f69b6c8235978c7451de60ecfdf9270f112c3e07d0a9d446fb361a74cae17","observation_id":"23141108-be84-401c-9425-b0ea07c0f062","resolution":{"observed_at":"2026-08-07T05:02:44.667779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.654337Z","title":"Funaudiollm: V oice understanding and generation foundation models for natural interaction between humans and llms,","venue":null,"work_id":"95ebec8f-42bf-4909-bcf5-1d299e58d23f","year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.409058Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:64109f20b2db27359cda548970be8776d6ccc98b507148f807ece3721f3aa7af","observation_id":"7a163650-412f-4999-b35a-f3a40cb377bf","resolution":{"observed_at":"2026-08-07T05:02:44.657772Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.644294Z","title":"Denes and E","venue":null,"work_id":"5f23778e-cb6f-4fe8-8da5-c5146014a1bb","year":1993},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.412025Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:aefbf7b30d0028dcf75f9dadcf0749f004f75382fd1d587ef820920d4380e993","observation_id":"bd0590be-51b2-47d1-91e2-d6935e2c57c8","resolution":{"observed_at":"2026-08-07T05:02:44.647512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.633404Z","title":"Listening while speaking: Speech chain by deep learning,","venue":null,"work_id":"76afebfe-2fcd-40b2-8b51-0a61ce84a565","year":2017},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.415140Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:0bea2505abc68c4e0b7380698efe96ca58215c465318806adecc1d36fe1967e2","observation_id":"97e7a478-4632-45aa-9323-fc8b3108d053","resolution":{"observed_at":"2026-08-07T05:02:44.636777Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.623193Z","title":"Machine speech chain with one-shot speaker adaptation,","venue":null,"work_id":"15391714-506b-4e32-bc88-75f64667edfc","year":2018},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.418056Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:012b7e560eff4d53567ec921848efdf6c704b4596c442144f01e2c03ba7b20de","observation_id":"89f55cb9-41a1-419c-ade4-b08e5efe1a43","resolution":{"observed_at":"2026-08-07T05:02:44.626338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.612367Z","title":"Speech chain for semi-supervised learning of japanese-english code-switching asr and tts,","venue":null,"work_id":"4912d6c6-7520-4ab2-80af-45ab4d2b0865","year":2018},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.421031Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:0c6493e124055e6ddf9c6bac890d93c070d26d49bdafb08f11043a49908b5b6d","observation_id":"49e5124b-0bfc-4865-82c1-0145612caaf9","resolution":{"observed_at":"2026-08-07T05:02:44.616372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.602007Z","title":"Montreal forced aligner [computer program],","venue":null,"work_id":"dcc2309a-13b1-4974-82e8-faceb64cf4ca","year":2017},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.424377Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:2f3642ee757efea249d7c4118e9bc880fe5aa9a2a1678de50193eaca7509c32b","observation_id":"8030fb06-d709-4e3f-8a2d-ce8a2a503d84","resolution":{"observed_at":"2026-08-07T05:02:44.605435Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.591568Z","title":"Fineweb2: A sparkling update with 1000s of languages,","venue":null,"work_id":"d5b2a7e3-f0e6-4845-a885-116ff8ea66c7","year":2024},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.427779Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:21a4511c848a99f73c9c10acf5691d2fd68d4d9b34a8bf3c2bc091ad652b20ab","observation_id":"fb1a391c-924d-4876-884e-39b2ab741760","resolution":{"observed_at":"2026-08-07T05:02:44.595146Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.431369Z","title":"Common voice: A massively-multilingual speech corpus,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.431369Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:a393042fa5b5271047d53d882b84e7ccd84312784f652836ab6c80b003f7567f","observation_id":"53f5a8a8-d7a1-4338-9232-31c84f2c0b5c","resolution":{"observed_at":"2026-08-07T05:02:44.431369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:02:44.572778Z","title":"Ascend: A spontaneous chinese-english dataset for code- switching in multi-turn conversation,","venue":null,"work_id":"b6d8820e-b4f8-43b1-9ff4-735f1da87210","year":2022},"citing_paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:44.434638Z"},"links":{"citing_paper":"/paper/2506.11130"},"observation_digest":"sha256:9854738feec3744313c06ff1c731aa37b7b8db682f6fa02435b5e7853f682dfe","observation_id":"3317f119-64c7-471b-8ada-0066f2c58191","resolution":{"observed_at":"2026-08-07T05:02:44.576759Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.11130","last_updated":"2025-06-16T15:47:41Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T04:54:43.738701Z","submitted_at":"2025-06-10T17:30:32Z","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data"},"reference_resolution":{"displayed":40,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":10,"verified_exact":1,"verified_fuzzy":29},"total_outbound_references":40},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 40 of 40 outbound references and 8 inbound Pith citation observations for arXiv:2506.11130."}