{"as_of":"2026-08-16T14:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:cbe4fa64561e611800eb8f6c856a659a2ea81fb36b5d75adf998cc0ea90b2811","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T19:25:16.279017Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T19:25:16.068793Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-15T19:25:16.387711Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"cited_work":{"arxiv_id":"2506.16741","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.16741","snapshot_observed_at":"2026-08-15T19:25:16.387711Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","venue":"eess.AS","work_id":"c896707b-4d8a-4b40-ba36-ef9c84c1e915","year":2025},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.068793Z"},"links":{"cited_paper":"/paper/2506.16741","citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:eae3b714251b28ce22d959f18c1396a0d8615c91c7e4b7063672d1c9f8d6791d","observation_id":"593623f1-8f2d-45a8-9830-3a5197ff8f60","resolution":{"observed_at":"2026-08-15T19:25:16.396542Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.16741/citation-record","integrity":"/paper/2506.16741/integrity","json":"/paper/2506.16741/citation-record.json","paper":"/paper/2506.16741"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:17.031062Z","title":"Among the various generative modeling approaches [4, 5, 6, 7, 8], ordinary differential equations (ODE)-based models [9, 10] have become strong solutions for outstanding TTS","venue":null,"work_id":"c7b80d57-2204-45e4-bb07-85b1a05fa496","year":null},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.062517Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:6b2d05408d92ca551e0f9dc2fb9e75e28470473be361ac813d166be5c106ac13","observation_id":"d486dca4-9f50-4712-bf1b-a0190ff8d43e","resolution":{"observed_at":"2026-08-15T19:25:17.038438Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"cited_work":{"arxiv_id":"2506.16741","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.16741","snapshot_observed_at":"2026-08-15T19:25:16.387711Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","venue":"eess.AS","work_id":"c896707b-4d8a-4b40-ba36-ef9c84c1e915","year":2025},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.068793Z"},"links":{"cited_paper":"/paper/2506.16741","citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:eae3b714251b28ce22d959f18c1396a0d8615c91c7e4b7063672d1c9f8d6791d","observation_id":"593623f1-8f2d-45a8-9830-3a5197ff8f60","resolution":{"observed_at":"2026-08-15T19:25:16.396542Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:17.012107Z","title":"TTS with Consistency Flow Matching For RapFlow-TTS, we follow the network design of Matcha- TTS [19] thanks to its fast and lightweight properties","venue":null,"work_id":"8e0628bc-852c-4e05-8cb8-26c785f99805","year":null},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.074634Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:13fd46bb3f3e2e919d4780095de8a88ef2c4adfea158802020bc796f9f1a3283","observation_id":"bbbd0fab-5983-46f1-a9c2-b96c6a62a09d","resolution":{"observed_at":"2026-08-15T19:25:17.017889Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.974132Z","title":"Experiment Setup DatasetTo verify RapFlow-TTS, we conducted experiments on the LJSpeech [27] and VCTK [28] datasets","venue":null,"work_id":"e7318584-ab0f-4521-ae34-c1e8ec51aac1","year":null},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.085787Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:f7af0d473dd12a50f4cb7f62a4eb83d26a9239290d5c9b5e19aa2243b1199bcd","observation_id":"a6a2586e-301c-4469-b7b8-1a3b338d7ba9","resolution":{"observed_at":"2026-08-15T19:25:16.980174Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.955512Z","title":"Based on consistency FM, RapFlow-TTS constructs a consistency model on a straight flow","venue":null,"work_id":"6a9dc60e-099c-4509-ae80-221ad122821f","year":null},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.093593Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:e5410f9e38fa68f613cd4ac158c1682c7f384c5a18ad7c6362a71693c468ec1b","observation_id":"54ffff0d-83be-46a3-9d38-8122c22ab6b8","resolution":{"observed_at":"2026-08-15T19:25:16.961714Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.937376Z","title":"Leaders in INdustry-university Co- operation 3.0","venue":null,"work_id":"e3084c8f-b48b-4c11-83f1-724d30ededfe","year":null},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.098991Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:36db71e36f37b5ad67d23d9619b67b5489f6cdc8a1bb3c768e6cb1ba0f82c2e3","observation_id":"699c773b-6398-48a2-b671-659511f9795e","resolution":{"observed_at":"2026-08-15T19:25:16.943498Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.804788Z","title":"FastSpeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":"87a7f0df-ca6f-4d9f-8069-5d42ae97f3b6","year":2021},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.140464Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:a5120c22add97e525b74bf2c037f35ec8716d378d293a12ba0d9c5652eccce6f","observation_id":"e9f6cfeb-ca7a-407a-b807-014de2a682b0","resolution":{"observed_at":"2026-08-15T19:25:16.810568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.919383Z","title":"Tacotron: Towards end-to-end speech synthesis,","venue":null,"work_id":"0bbff7a7-307c-4c52-a69f-7ca503b832a5","year":2017},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.105261Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:72a13179b6686ec0c79de9ba0de1edbcac3178cfb4c5d4e95857dc721efe1f58","observation_id":"fd254422-2441-4962-a843-9cccfb1dbe1a","resolution":{"observed_at":"2026-08-15T19:25:16.925240Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.901070Z","title":"Natural TTS synthesis by conditioning Wavenet on mel spectrogram predic- tions,","venue":null,"work_id":"443a78f2-3409-4830-abb3-eb44066c7f14","year":2018},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.110727Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:88619a10c0b573071a534d08e180bf8a65f2b609eec6b1804089316023814286","observation_id":"dba2e49d-52c7-4582-a796-774698b8f6dd","resolution":{"observed_at":"2026-08-15T19:25:16.906778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.882681Z","title":"Glow-TTS: A generative flow for text-to-speech via monotonic alignment search,","venue":null,"work_id":"b3a74af0-6c96-4ebf-be5c-156d27f30e7b","year":2020},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.117180Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:a5edb45e6f8bf2e5fdb0d687047a76a4bae03ae3ad0518499f6e57054b1af282","observation_id":"bdebd58b-c17c-4eff-a7a1-f86753d04681","resolution":{"observed_at":"2026-08-15T19:25:16.888252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.864629Z","title":"FastSpeech: Fast, robust and controllable text to speech,","venue":null,"work_id":"28ffc418-b7d2-463e-a820-1114dd489471","year":2019},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.122742Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:ba808128415c64843f4153d60976da4a52e30f38e11bbcd6d14a016ba318ce25","observation_id":"5c742b8d-76ff-4509-8b1a-83c8c04dbd45","resolution":{"observed_at":"2026-08-15T19:25:16.870196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.845854Z","title":"Flow- TTS: A non-autoregressive network for text to speech based on flow,","venue":null,"work_id":"5b534d15-0589-421c-86b4-9f4bbd9666f4","year":2020},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.128819Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:45132392d66a7631d7062e15152b5926d55ee60274ffc4dba9668efaa2e75c38","observation_id":"976da085-10a7-488a-9c25-caed4939adf1","resolution":{"observed_at":"2026-08-15T19:25:16.851868Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.825910Z","title":"Neural speech synthesis with Transformer network,","venue":null,"work_id":"d8600abf-1996-4e40-9b19-911fadb169fd","year":2019},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.134796Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:df023a06ba417f9b8957e6a60eb0218769ee7188c3fc2c6db2335cd5e4b904c0","observation_id":"8fce1e51-50ad-4fca-8de1-df64af0459d5","resolution":{"observed_at":"2026-08-15T19:25:16.831788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.694150Z","title":"Improving and generalizing flow-based generative models with minibatch optimal transport,","venue":null,"work_id":"4599008d-5823-4beb-86fb-adfe3630d402","year":2024},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.177931Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:42db69d45120d80be631bcd45eaf5cd56bce7fbf74a94b4ab9a73a9e1fd657eb","observation_id":"c327802d-6057-4342-be04-de7691a9ce0d","resolution":{"observed_at":"2026-08-15T19:25:16.700215Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.785854Z","title":"Diff-TTS: A denoising diffusion model for text-to-speech,","venue":null,"work_id":"0049c984-0d2c-4f58-ae7a-6ec018db35fb","year":2021},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.145662Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:684a0cbd13dc986e69d22a03fcd8b0131510f1aa0ab988e7fc4a994a16b3a9f1","observation_id":"995baa21-53ca-4407-a627-60b6a2ece62d","resolution":{"observed_at":"2026-08-15T19:25:16.791971Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.766720Z","title":"Grad-TTS: A diffusion probabilistic model for text-to-speech,","venue":null,"work_id":"0f35e2c0-7069-41fa-87a4-61f419326186","year":2021},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.150846Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:16ebc49fbd1a32418125b0b732cd08a991fb83bd760def722e5cf078338d7f24","observation_id":"8043b4b9-7e9f-498a-8694-7f429320dc4f","resolution":{"observed_at":"2026-08-15T19:25:16.772623Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19135","last_updated":"2024-06-27T12:39:55Z","snapshot_observed_at":"2026-08-16T13:39:05.618443Z","submitted_at":"2024-06-27T12:39:55Z","title":"DEX-TTS: Diffusion-based EXpressive Text-to-Speech with Style Modeling on Time Variability","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.19135","snapshot_observed_at":"2026-08-15T19:25:16.156489Z","title":"DEX-TTS: Diffusion-based EXpressive Text-to-Speech with style modeling on time variability,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.156489Z"},"links":{"cited_paper":"/paper/2406.19135","citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:dd2791818b0401ccc3db17093f96fef63817d6986215e1eb68e07b709caba613","observation_id":"722e685c-fa66-4cb8-ad44-b8e8ee5cb6ef","resolution":{"observed_at":"2026-08-15T19:25:16.156489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.746200Z","title":"Score-based generative modeling through stochas- tic differential equations,","venue":null,"work_id":"9a374eff-b0d7-460c-bc35-e6327728cc86","year":2021},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.162101Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:f74527a12f66e0765a648ef49f807bd32d528239ba7e5b7702b693016d675a2e","observation_id":"df7a04e7-53d9-4c99-907f-0577da2c9060","resolution":{"observed_at":"2026-08-15T19:25:16.753786Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.726160Z","title":"Elucidating the design space of diffusion-based generative models,","venue":null,"work_id":"fbfe404a-62c0-4de9-b11f-7f7555015ad8","year":2022},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.167228Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:039c8d5cc187203f926b8008a2c947e30796ccbdeadb2696fd2643c3da8eaa68","observation_id":"dc2fcb33-5854-4539-aa37-9a4f4fbf7a73","resolution":{"observed_at":"2026-08-15T19:25:16.731702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.172938Z","title":"Flow matching for generative modeling,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.172938Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:7afb06d3aa55e447efdb96e4af5daf7b7d9b972b1e21bfa6b2da98255a026eb8","observation_id":"6493510d-ab7a-4cd9-95ed-9a97eb2ad3f7","resolution":{"observed_at":"2026-08-15T19:25:16.172938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.577324Z","title":"Improved techniques for training con- sistency models,","venue":null,"work_id":"3311eb81-ad74-47be-b68b-2dfab3a34aca","year":2024},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.217422Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:0022fdb5b9f96625d15e41b0ea3e9deac670f102282ae65f431fd12303e40a4f","observation_id":"d55919d4-6992-4653-bb69-c5a22e9ac050","resolution":{"observed_at":"2026-08-15T19:25:16.582991Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.675218Z","title":"V oiceflow: Efficient text-to-speech with rectified flow matching,","venue":null,"work_id":"eee76df1-03f2-4b3b-8103-707ec6c523c6","year":2024},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.183380Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:b05585216e5cf9f0bdf1a78ce0636970c67b226587d96dfbeff06221819fe9a7","observation_id":"48a809d2-8332-4c7a-ac3b-96f76c0fe25e","resolution":{"observed_at":"2026-08-15T19:25:16.680974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.655585Z","title":"Flow straight and fast: Learning to generate and transfer data with rectified flow,","venue":null,"work_id":"2c582690-9abf-4e50-9ba2-934b341fbd26","year":2023},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.188683Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:071ab51945d180613457e01412f1d38ec7e9bf932645681552a129d6cee37bb6","observation_id":"2d9554aa-0bae-47ec-865e-9a4c012ab9ed","resolution":{"observed_at":"2026-08-15T19:25:16.661995Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.634527Z","title":"Consistency models,","venue":null,"work_id":"33b0cd23-09a7-4589-9643-b5ead6943f97","year":2023},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.194305Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:5d90356375377f8bec995087bafa49a60911dc3799e2f1a26a23829669282b95","observation_id":"90a24090-99ed-4bcb-9c0c-e56c0a37c4d5","resolution":{"observed_at":"2026-08-15T19:25:16.641652Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.993477Z","title":"Furthermore, we extend it to multi-segment adversarial learning for consistency FM","venue":null,"work_id":"7d9721c5-454f-4596-a134-0c39ebf78031","year":null},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.080395Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:c5f39e4fd4a922dc02a95ab0aaaabf53cf926fb29ae31191f496f21e4c8de410","observation_id":"52f71ff1-a622-4153-808f-dfb53bcdc30b","resolution":{"observed_at":"2026-08-15T19:25:16.999691Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.615735Z","title":"Como- speech: One-step speech and singing voice synthesis via consis- tency model,","venue":null,"work_id":"de03e9a5-169e-4064-9b66-cff042ae25a1","year":2023},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.200560Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:cbd84b9a8d115b060e50fac665e4e2b5590ac99e27c388665b6926c2b3a8f8cf","observation_id":"c7338edb-b412-462b-a429-de0675e1cfbd","resolution":{"observed_at":"2026-08-15T19:25:16.621207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.597144Z","title":"Matcha-TTS: A fast tts architecture with conditional flow match- ing,","venue":null,"work_id":"6d981b51-6f2b-4cf7-acf1-7be144a48816","year":2024},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.205514Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:336ab6ec452e10204dbc09980a1abc525b86266697c1dbd663c8557abc0d69ff","observation_id":"d4c043a6-fc1a-4a6f-8202-85cb851d8380","resolution":{"observed_at":"2026-08-15T19:25:16.603200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.02398","last_updated":"2024-07-02T16:15:37Z","snapshot_observed_at":"2026-08-16T13:37:41.452512Z","submitted_at":"2024-07-02T16:15:37Z","title":"Consistency Flow Matching: Defining Straight Flows with Velocity Consistency","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.02398","snapshot_observed_at":"2026-08-15T19:25:16.211733Z","title":"Consistency flow matching: Defining straight flows with velocity consistency,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.211733Z"},"links":{"cited_paper":"/paper/2407.02398","citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:8e64bb9e4db9bee0dd26b1a93d5f90ba90ffd573fe40050a4b6774f0fb1313e6","observation_id":"4b72fbb8-b8b1-4434-8e08-59d92c5ee09a","resolution":{"observed_at":"2026-08-15T19:25:16.211733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.557591Z","title":"Simple reflow: Improved techniques for fast flow models,","venue":null,"work_id":"f3efddb6-a3ae-413d-9353-7957587e8841","year":2025},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.223150Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:d5090362883cc5d8517c23ea816bd9b5bd36b9f77f5061eeff2fa12050c9477e","observation_id":"dded721e-13a0-4211-8532-5b0da25b273f","resolution":{"observed_at":"2026-08-15T19:25:16.563970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.534898Z","title":"Improving the training of rectified flows,","venue":null,"work_id":"f03c41dc-249f-4980-92be-91205105b51a","year":2024},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.229560Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:dff4c10733c6818f8ed264a64ee33e88751c58ac92fe4efb68e5e860532fd21d","observation_id":"ff5926ac-a597-41d3-ba65-b22b2c65dc7d","resolution":{"observed_at":"2026-08-15T19:25:16.542975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.236398Z","title":"Least squares generative adversarial networks,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.236398Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:b21c343eed65dbcb6fa25f1677e6cefc20a01a6302cfb6a28c06c75da343300f","observation_id":"afab52f4-7fc4-422b-a7c3-65382f413b3b","resolution":{"observed_at":"2026-08-15T19:25:16.236398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.499774Z","title":"Improved techniques for training gans,","venue":null,"work_id":"12714e1a-79ff-41b1-86f4-5f9ed7483e86","year":2016},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.241849Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:572c7ec46828ef3a76f318116b80c63ca9136d9a3a445c84419fe2d36288b5ec","observation_id":"b0001c02-e9ca-4699-bf99-c6797bf7966d","resolution":{"observed_at":"2026-08-15T19:25:16.505899Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.15439","last_updated":"2023-11-20T04:31:13Z","snapshot_observed_at":"2026-08-13T15:33:42.741748Z","submitted_at":"2022-05-30T21:34:40Z","title":"StyleTTS: A Style-Based Generative Model for Natural and Diverse Text-to-Speech Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.15439","snapshot_observed_at":"2026-08-15T19:25:16.247483Z","title":"StyleTTS: A style-based generative model for natural and diverse text-to-speech synthesis,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.247483Z"},"links":{"cited_paper":"/paper/2205.15439","citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:7ace9d15de2b1bd24f2e123bc5a219f84715ab6ed6325348d8310426d8f332c6","observation_id":"cedd6373-444c-41af-8de8-ad01e718ea8f","resolution":{"observed_at":"2026-08-15T19:25:16.247483Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.253448Z","title":"The LJ speech dataset,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.253448Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:b122d0250e6358e09a44516bed6a4d0a499c2ddd24f1803d156b9cb16ed216b0","observation_id":"24b5e92a-d8b7-4661-894d-0cb830e57735","resolution":{"observed_at":"2026-08-15T19:25:16.253448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.466221Z","title":"CSTR VCTK cor- pus: English multi-speaker corpus for CSTR voice cloning toolkit (version 0.92),","venue":null,"work_id":"6aab9bd3-1265-4c40-af68-a5bd5e7e36b1","year":2019},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.261638Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:e5014329ac03e26d43a495a36bdbe1b30d2cba7a53a3f418576d3d93ba880fb7","observation_id":"1446dc55-bcff-4332-b445-310469b509c2","resolution":{"observed_at":"2026-08-15T19:25:16.472218Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.266581Z","title":"HiFi-GAN: Generative adversarial networks for efficient and high fidelity speech synthesis,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.266581Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:2928c508432926ec5c598c422127926781d858133c9e10413f635a2fb848f3c2","observation_id":"0d8a774b-870b-49f6-b8e8-4a337ee34cee","resolution":{"observed_at":"2026-08-15T19:25:16.266581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.272210Z","title":"Robust speech recognition via large-scale weak su- pervision,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.272210Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:f39448339934682212ac923c79010531b35bf2e89b420e78c6ef4dd081e808db","observation_id":"65982994-dec7-4045-ae05-19de6d5b7fe8","resolution":{"observed_at":"2026-08-15T19:25:16.272210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:25:16.410970Z","title":"NISQA: A deep CNN-self-attention model for multidimensional speech quality prediction with crowdsourced datasets,","venue":null,"work_id":"a0b10e81-69c0-411f-8624-017c257e6a97","year":2021},"citing_paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T19:25:16.279017Z"},"links":{"citing_paper":"/paper/2506.16741"},"observation_digest":"sha256:bd4d8cadc08fad8d65e3c89c0a009c1f615aee1774b6a7fea4f88dfd4b82eda5","observation_id":"63db88d6-b89b-48be-97ef-6259e7294ae7","resolution":{"observed_at":"2026-08-15T19:25:16.417869Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.16741","last_updated":"2025-06-20T04:19:29Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-15T19:17:23.654198Z","submitted_at":"2025-06-20T04:19:29Z","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":8,"verified_exact":1,"verified_fuzzy":29},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 1 inbound Pith citation observation for arXiv:2506.16741."}