{"as_of":"2026-08-05T09:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:23ddaad88de7f5011846f1669a97b331fde1fe43ac326e537802760c19a52dd6","coverage":[{"denominator":66,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":66,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-15T11:56:13.914121Z","state":"measured"},{"denominator":67,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":67,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T15:46:58.010112Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-05-11T09:51:00.769273Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"cited_work":{"arxiv_id":"2603.14267","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.14267","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","venue":"cs.CV","work_id":"fee41a5a-328d-4d4e-b980-be95ffce9f61","year":2026},"citing_paper":{"arxiv_id":"2604.12292","last_updated":"2026-04-14T05:03:57Z","snapshot_observed_at":"2026-07-06T23:00:32.607699Z","submitted_at":"2026-04-14T05:03:57Z","title":"CoSyncDiT: Cognitive Synchronous Diffusion Transformer for Movie Dubbing","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T15:46:58.010112Z"},"links":{"cited_paper":"/paper/2603.14267","citing_paper":"/paper/2604.12292"},"observation_digest":"sha256:02ae54d58e8c2cf85259c0ac12e5ae41c4e737f52309807c5911c6950b1af04b","observation_id":"886aca88-63d5-4480-afe1-cc6036298bfe","resolution":{"observed_at":"2026-05-11T09:51:00.772682Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2603.14267/citation-record","integrity":"/paper/2603.14267/integrity","json":"/paper/2603.14267/citation-record.json","paper":"/paper/2603.14267"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","venue":null,"work_id":"c821a402-d6bf-4c26-b0ea-d7117eabea49","year":2020},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:73993110f684bf9e15931cd900501f9433ca5caac3625622a1fd2dd350996c80","observation_id":"9d5f1bba-507c-4995-b77c-a2db99d50867","resolution":{"observed_at":"2026-05-15T12:00:00.318239Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"V2c: Visual voice cloning","venue":null,"work_id":"b49d2da6-2457-4b77-a30a-d6e4d9a042c3","year":2022},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:d38f4d8be685758bed1e0917f79485cf3e02ed96e1e4618e18dd284bce0523a2","observation_id":"1d678fc8-8c2e-4127-bdd9-d8e39793d0f3","resolution":{"observed_at":"2026-05-15T12:00:00.263338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":null,"work_id":"89046935-5ae5-4267-82d0-3eac8c2c15ff","year":2024},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:3571ce506779024d6991002a111c1ec92c3e4a3de091bdf9f42d0fff800ae6f6","observation_id":"8a06e516-a4b9-40c4-ac2b-67d1c6f02cfb","resolution":{"observed_at":"2026-05-15T12:00:00.315683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Neural codec language models are zero-shot text to speech synthesizers","venue":null,"work_id":"3ee791a7-331f-4bb5-bc39-7bba751f0f42","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:e24f81357dcf9a09233d1a1bca85da794c45e281c25de6e51403a9709ab69f45","observation_id":"0b9d66b4-0c1b-478f-8411-b3b0341ad95a","resolution":{"observed_at":"2026-05-15T12:00:00.320855Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"F5-TTS: A fairytaler that fakes fluent and faithful speech with flow matching","venue":null,"work_id":"4d92fb0f-7079-4751-b919-4fb1d44946b8","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:1100b040d594941db6d9e6c68c72372cca30599488ac74ffe4d401ff6970e9d7","observation_id":"9a249c86-d129-439a-99a1-9e4e73fb4754","resolution":{"observed_at":"2026-05-15T12:00:00.288345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Intelligible lip-to-speech synthesis with speech units","venue":null,"work_id":"d2ea64d2-1813-4571-8fb5-daba27837185","year":2023},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:de30c0917fd3196f4ad649667d999250233f06d2ec8b93bb474863da799e271a","observation_id":"78267e3a-602d-4195-a58e-580949862523","resolution":{"observed_at":"2026-05-15T12:00:00.328390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Aligndit: Multimodal aligned dif- fusion transformer for synchronized speech generation","venue":null,"work_id":"b3817742-3991-492c-b5c4-ea8ff8259523","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:2a86727b1d6df433ad34c6a9eca6ca2a8a587ac03026c3b44748a57a47a59480","observation_id":"24c5ffae-93be-4c96-ac7c-116228609ef0","resolution":{"observed_at":"2026-05-15T12:00:00.274474Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Accelerating Diffusion- based Text-to-Speech Model Training with Dual Modality Alignment","venue":null,"work_id":"c52be7aa-aece-4248-ad15-ef21f340af55","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:c77d43c840d4788e50a3f5c2d81f96ea9d4b3ece8020e79e73a64e7c47d80516","observation_id":"3ff8c2c0-a1eb-4df4-95af-a7471d34c2dc","resolution":{"observed_at":"2026-05-15T12:00:00.267812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Reducing f0 frame error of f0 tracking algorithms under noisy conditions with an un- voiced/voiced classification frontend","venue":null,"work_id":"0e18e65c-664a-4b60-8b6e-be8f34d6b350","year":2009},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:9c1e1fce928411cb68fe567c99d6148b6f61a7ce3dc872572e1bf122180978ad","observation_id":"aa7c9b87-bf90-4dc2-848d-dc3346e30468","resolution":{"observed_at":"2026-05-15T12:00:00.270998Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Out of time: au- tomated lip sync in the wild","venue":null,"work_id":"9ea572d8-4d76-4db1-be76-189b5dfa344e","year":2016},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:fd6515b120c2a09b75ef2b2f827abfb9947c7adf11784bf20f10a37e56144d6e","observation_id":"6113a4be-2e74-42ec-b59e-4a8f894d0367","resolution":{"observed_at":"2026-05-15T12:00:00.319183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning to dub movies via hierarchical prosody models","venue":null,"work_id":"a8e7b97a-5465-4d81-ae23-1d1025160ae6","year":null},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:d9de6cd1917e7d3834a58d64a131020f7a52917c1df71b7d4d8d9a05e868ac53","observation_id":"65516fd5-6a0e-44db-99b9-b14de03b9c53","resolution":{"observed_at":"2026-05-15T12:00:00.341942Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"StyleDubber: Towards multi- scale style learning for movie dubbing","venue":null,"work_id":"2c6f17a0-b570-43b6-8be3-f84fc73a242e","year":2024},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:5a44b0a32a0ddd4ec2475722a4993e68261f3498e4f5a841818322e5ddad6af2","observation_id":"ecbf7438-009c-4c50-9b91-65490977c615","resolution":{"observed_at":"2026-05-15T12:00:00.334665Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Emodubber: Towards high quality and emotion con- trollable movie dubbing","venue":null,"work_id":"cf05b410-4a49-483a-918c-84229ea474fa","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:846d979a757a4dc598006213e1106bae9111ffce3db4e309e3a466557bdb47a8","observation_id":"d0dd85ca-9e14-46ab-b4a9-62df08bcc87a","resolution":{"observed_at":"2026-05-15T12:00:00.364485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"An audio-visual corpus for speech perception and automatic speech recognition.The Journal of the Acoustical Society of America, 120(5):2421–2424","venue":null,"work_id":"8ccdd660-1bce-4ed9-aaab-085d36a2bdec","year":2006},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:a7560e4989531a3ff24dc63947a34541da1de18ac5bd7ee77cd7a5b36bcceb43","observation_id":"0b3d3851-40f9-40ce-b885-75240137e53c","resolution":{"observed_at":"2026-05-15T12:00:00.229695Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sigmoid- weighted linear units for neural network function approx- imation in reinforcement learning.Neural networks, 107: 3–11","venue":null,"work_id":"85bcd9f6-b8b3-4316-a148-993ed2088513","year":2018},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:679b90b7f95b219c937b489f0bc23c832a5186bafed761ec7d81be07739b52d0","observation_id":"610db2db-cf00-4f9a-8785-5b95df970a71","resolution":{"observed_at":"2026-05-15T12:00:00.341579Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"E2 tts: Embarrassingly easy fully non-autoregressive zero-shot tts","venue":null,"work_id":"91d6bc5a-6aef-42f8-a05a-93711d1d8087","year":null},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:d1a8468779471d29a022e7d9817b81b968c716672bd1e43cfc164a7f2c3016a2","observation_id":"eded40d4-19d1-4aa7-8c13-607fbaed4df8","resolution":{"observed_at":"2026-05-15T12:00:00.345188Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"LLaMA-omni: Seamless speech interaction with large language models","venue":null,"work_id":"53bc568c-1565-4879-ba00-efcdef60a8fb","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:ba23477e78248185a86289f15654ab3111f2b8baae381173db0b6fe23122da5f","observation_id":"05cec627-2750-4cf7-b4ec-a65fbff14f5e","resolution":{"observed_at":"2026-05-15T12:00:00.225863Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"877f8832-a608-4fa5-ba1e-47ad35487664","year":2024},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:72eb2a3029aac417befefd46f6c708caad69bca87b8056a062dd5a3c64dba18a","observation_id":"7af4c8a5-b4c5-44e7-8cc5-2634a55c9f33","resolution":{"observed_at":"2026-05-15T12:00:00.237377Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"V oiceflow: Efficient text-to-speech with rectified flow match- ing","venue":null,"work_id":"bd69d2b5-9eb5-460e-9cb9-70cf7e53ce5b","year":2024},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:adb57419363635846a827b6e089660deb2f1d373818a86f753f0c4edd86406d7","observation_id":"9230ce63-01fe-45c3-a14b-00f7964073b3","resolution":{"observed_at":"2026-05-15T12:00:00.370565Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07855","last_updated":"2024-06-12T04:09:44Z","snapshot_observed_at":"2026-08-05T04:07:33.037081Z","submitted_at":"2024-06-12T04:09:44Z","title":"VALL-E R: Robust and Efficient Zero-Shot Text-to-Speech Synthesis via Monotonic Alignment","version":1},"cited_work":{"arxiv_id":"2406.07855","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.07855","snapshot_observed_at":"2026-07-03T23:19:03.868044Z","title":"Vall-e r: Robust and efficient zero-shot text-to-speech synthesis via monotonic alignment","venue":null,"work_id":"d46897fa-6b0f-4803-b02b-59b87aee1d4b","year":2024},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"cited_paper":"/paper/2406.07855","citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:fbcd983fe59461b99749271ca1bf59b315efede76c305d3f34d4d2a419a279f2","observation_id":"972ee4d0-1303-4922-bdb3-2f0576d74bae","resolution":{"observed_at":"2026-05-15T11:59:59.585731Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Boosting large language model for speech synthesis: An empirical study","venue":null,"work_id":"94ad48f1-32ca-4174-8187-8d00910053bc","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:7c7be423f068a1c57cb825f292b77b3b15f07ea506954a520b3ff351c1265a75","observation_id":"be3df557-aabd-4dfb-aa11-22010fe593a4","resolution":{"observed_at":"2026-05-15T12:00:00.331174Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1606.08415","last_updated":"2023-06-06T01:53:32Z","snapshot_observed_at":"2026-07-06T05:01:27.910364Z","submitted_at":"2016-06-27T19:20:40Z","title":"Gaussian Error Linear Units (GELUs)","version":5},"cited_work":{"arxiv_id":"1606.08415","doi":"10.18653/v1/n19-1122","metadata_source":"pith","pith_arxiv_id":"1606.08415","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gaussian Error Linear Units (GELUs)","venue":"cs.LG","work_id":"0466fd22-03a1-4a61-af0a-a900e77bb023","year":2016},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"cited_paper":"/paper/1606.08415","citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:a500395099f520f650100322ce47289fb1a857ba49fb03a8f63f741289de646a","observation_id":"ce3bc253-7c51-46bf-b025-60570bd53054","resolution":{"observed_at":"2026-05-15T11:59:59.605778Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"OZSpeech: One-step zero-shot speech synthesis with learned-prior-conditioned flow matching","venue":null,"work_id":"0be4030a-b082-46f8-a460-fd07881a8645","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:6b028d37d4dd3a6b4b6a38f0a683e534da0d8cb43e38f6743dfa315124d6b992","observation_id":"eda4591f-69d0-4a8a-a2a1-d51027a49c27","resolution":{"observed_at":"2026-05-15T12:00:00.409904Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Neural dubber: Dubbing for videos according to scripts","venue":null,"work_id":"0b942341-8286-427e-a7e0-481f566866c2","year":2021},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:5dc32931b487b2f76aeb0238bda53bb78a61206ec2756f7434a6ab3b094f5674","observation_id":"83c8e916-1ff7-41bc-a4bf-56e67d60473b","resolution":{"observed_at":"2026-05-15T12:00:00.419240Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hunt and A.W","venue":null,"work_id":"b194c54a-0d8b-4043-977b-78c83dc7b11e","year":1996},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:d5d77d5c1e7890e68d0aed59ee23352553ae647d85c5107fa99834bb7ca4c947","observation_id":"8d86afa4-e82d-4cd7-8dd9-fe2d248850cf","resolution":{"observed_at":"2026-05-15T12:00:00.378570Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"MobileSpeech: A fast and high-fidelity frame- work for mobile zero-shot text-to-speech","venue":null,"work_id":"303dbb85-7e94-4e67-a98f-76b7c4667fa3","year":2024},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:fcb5fff4e9474357dcdb9dc6989e307d442f9aca333ec59ae4fb9a7488023f60","observation_id":"b13465eb-c358-47ce-93d0-ffc7f18cfc66","resolution":{"observed_at":"2026-05-15T12:00:00.338292Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"NaturalSpeech 3: Zero-shot speech synthesis with factor- ized codec and diffusion models","venue":null,"work_id":"983438f6-f37c-4009-917d-a56b508d8c00","year":2024},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:68c8a98ca38d5ab62194ee9f814c201f3e9b3d7940202b5b649db72c5dbef075","observation_id":"33523617-7809-41ba-9b6a-61afe36c93b1","resolution":{"observed_at":"2026-05-15T12:00:00.249282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Zet-speech: Zero-shot adaptive emotion-controllable text-to- speech synthesis with diffusion and style-based models","venue":null,"work_id":"6e9b0202-65e9-42a8-80b2-873d3aead8f4","year":2023},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:8688232cc6666e508315c0c1267555b9021a740b16355718cccf8f448baf437c","observation_id":"0707335d-1c50-4199-bbf3-b7666a8f69f4","resolution":{"observed_at":"2026-05-15T12:00:00.347000Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Glow-tts: A generative flow for text-to-speech via monotonic alignment search","venue":null,"work_id":"07614c7b-a15d-4ae5-a463-a3cc99a7844f","year":2020},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:bc40bf953df51e75a256093523405eb46c88966a9ed38c4db459fce1afc91634","observation_id":"91dd46d8-7084-42e2-888b-aab5cad29b4a","resolution":{"observed_at":"2026-05-15T12:00:00.278006Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Shih, Rohan Badlani, Joao Felipe Santos, Evelina Bakhturina, Mikyas T","venue":null,"work_id":"ee4aa9c9-9aae-4656-85d2-cc3d43928b1f","year":2023},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:5d1a6765715d69ab8336f09e1d4910aa0b04245db673138e6ee7a998288bdded","observation_id":"013720c2-a86e-4ca4-94dd-56a97aeed683","resolution":{"observed_at":"2026-05-15T12:00:00.284703Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"V oicebox: Text- guided multilingual universal speech generation at scale","venue":null,"work_id":"461e120d-371f-4455-9411-50a32c9f95f3","year":2023},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:87978ff885dc27ab08888cec9144ca20adeacc8bfe36c893fb75a3bb19173d0b","observation_id":"92891c00-3e5c-433a-aee5-477f5062b300","resolution":{"observed_at":"2026-05-15T12:00:00.244587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"DiTTo-TTS: Diffusion transformers for scalable text-to-speech without domain-specific factors","venue":null,"work_id":"6279b162-5b4a-4503-a1b5-aebe8a2682d7","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:ef7e83124f43eb7db5df16a80e2f94333c1f658c8c2ac396e919ed7ae24a6301","observation_id":"fbd66383-48ca-4d87-bbb1-ee6403c2eca6","resolution":{"observed_at":"2026-05-15T12:00:00.288898Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Decoupled weight de- cay regularization","venue":null,"work_id":"40928b59-2bc3-49c3-a7e3-cea897f1b2dc","year":2019},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:49db7d95c089fbcca599a19dadd6ef37549097738efe31b73ff8f51240212d2c","observation_id":"5c43cf4b-cb38-4815-9f3b-e93c03285115","resolution":{"observed_at":"2026-05-15T12:00:00.361770Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Monotonic multihead attention","venue":null,"work_id":"27bb95cd-8408-4210-8624-597a052e754d","year":2020},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:979d1d5c74c7492f8e71267864d8583234595b25428022c33ac828528229721c","observation_id":"0db26fd0-7396-452f-b1a8-3d1895eed41a","resolution":{"observed_at":"2026-05-15T12:00:00.335784Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Montreal forced aligner: Trainable text-speech alignment using kaldi","venue":null,"work_id":"72e0310e-a9be-46c4-a804-95942123307f","year":2017},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:598264364da2f5103acb5ad89c029a9ec5f20cc7bda34edeb3fbe7d5209d2a33","observation_id":"72858e84-e149-4e9f-a9f1-931307c2d965","resolution":{"observed_at":"2026-05-15T12:00:00.260249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Matcha-tts: A fast tts architecture with conditional flow matching","venue":null,"work_id":"7fe4b789-db23-403a-a3c6-2419ed83006a","year":2024},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:324dfb41b5696d0831948d4c7bd5c50ca462ab5bbf57170847ae917aea73fb3a","observation_id":"a251bfc1-96a7-4048-8339-e99a18161037","resolution":{"observed_at":"2026-05-15T12:00:00.321390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Meng, and Furu Wei","venue":null,"work_id":"9823eb1a-dfff-4d9b-a3bb-780795fa6912","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:7e6bda2251eb3db6c0bb71d2b10456de4d4e18ca903daca08242dc995882ee14","observation_id":"5ae73557-32a8-4416-8004-af9aa69e2c3f","resolution":{"observed_at":"2026-05-15T12:00:00.323268Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08681","last_updated":"2020-08-13T05:42:12Z","snapshot_observed_at":"2026-07-06T08:16:16.271286Z","submitted_at":"2019-08-23T06:22:06Z","title":"Mish: A Self Regularized Non-Monotonic Activation Function","version":3},"cited_work":{"arxiv_id":"1908.08681","doi":null,"metadata_source":"pith","pith_arxiv_id":"1908.08681","snapshot_observed_at":"2026-07-08T11:24:54.882129Z","title":"Mish: A self regularized non-monotonic activation function","venue":"cs.LG","work_id":"f314c3b6-cec7-4246-8006-cbed6b9840d3","year":2019},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"cited_paper":"/paper/1908.08681","citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:cd3e283a3d6e8f480fce78360c66ba1ac28b096ee37dfb8d41d225369d8e6431","observation_id":"3c2b4ae7-6690-47b1-a932-130221a83ab8","resolution":{"observed_at":"2026-05-15T11:59:59.601393Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5fe9c58a-4310-496c-9ecb-2afcbb5444e0","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:dfaa51dcf535fb83659addec34415dd65ef2c4786e775378f8fe7d995f6ba998","observation_id":"7d0661db-97be-4de2-8a1f-81cc01666265","resolution":{"observed_at":"2026-05-15T12:00:00.373497Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"g2pe","venue":null,"work_id":"7f95b6c8-4a25-4f43-b412-996c061d6cd5","year":2019},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:67e4acf01dea851a7abeac65184a05ebbfd1c71ea99d3115e4dc5bcffb9f9baa","observation_id":"9e85afdf-2bbd-4be9-91e7-3fbad0df55e5","resolution":{"observed_at":"2026-05-15T12:00:00.376057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T05:14:35.715460Z","title":"Scalable diffusion models with transformers","venue":null,"work_id":"292e4202-5927-473f-b73e-1d6883498f14","year":null},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:c3e0f4466e360fe11bb45751d363c9f4d23fde69360bb92d6fc3ab2c35e1e6b4","observation_id":"a51c1974-851f-4233-808c-5b6c7f5fc9d6","resolution":{"observed_at":"2026-05-15T12:00:00.382071Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"VoiceCraft: Zero-shot speech editing and text-to-speech in the wild","venue":null,"work_id":"64c0188d-8435-45dd-b253-9315978d1f55","year":2024},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:a55d3584f81afc34a0343e870755993aec9e1d0cc758d32ffcd55b890e220fa7","observation_id":"0aa3e1f0-945f-4ba2-873f-fcac30605741","resolution":{"observed_at":"2026-05-15T12:00:00.388183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"397f681c-effd-41c5-9ac6-26294d550eed","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:5dc6805f95db31d51e6e04b0ebc7e1f69a29c697d43fd5159be11b31914540b1","observation_id":"22c663d5-c5e6-4bbf-aec6-6b26cf43b5e9","resolution":{"observed_at":"2026-05-15T12:00:00.391889Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Speech resynthesis from discrete disentangled self-supervised representations","venue":null,"work_id":"70a343e9-83fe-469e-805b-dbf2bb3f4b11","year":2021},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:dbe670bdd80360f1ffc811667a2c7d6aa6ddc9a3d08b7523fa00c6a14245287a","observation_id":"b76406a8-6833-4157-8000-65cd8ea8ab7c","resolution":{"observed_at":"2026-05-15T12:00:00.416611Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Robust speech recog- nition via large-scale weak supervision","venue":null,"work_id":"e266454e-18a8-4428-96c2-a0fd5087a0a0","year":2023},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:899591eb99aa1e8c72c3e262b81a924d3ab7b7ea4f2098e05c85931e390e8796","observation_id":"269ba7b0-f5e4-4e7d-bc8a-63d08b6f3fc6","resolution":{"observed_at":"2026-05-15T12:00:00.422015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fastspeech: Fast, robust and con- trollable text to speech","venue":null,"work_id":"a25d348d-e6c1-421b-b52b-84f371806416","year":2019},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:93faa1af743e071ee43c629d5418c053bebf45b451821d51bf847a4911f703cc","observation_id":"1a56d666-ccce-4957-adde-0f7f44932a82","resolution":{"observed_at":"2026-05-15T12:00:00.404267Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fastspeech 2: Fast and high-quality 10 end-to-end text to speech","venue":null,"work_id":"6addcb1a-6eed-40e8-b0d2-12c573d0cb33","year":2021},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:234589df21e0b0fe468ced622701686f561b217ddbc42315c546e52b02f528f1","observation_id":"feda58f0-07e5-433c-98eb-3c4618056172","resolution":{"observed_at":"2026-05-15T12:00:00.398848Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Weiss, Mike Schuster, Navdeep Jaitly, Zongheng Yang, Zhifeng Chen, Yu Zhang, Yuxuan Wang, Rj Skerrv-Ryan, Rif A","venue":null,"work_id":"9a695ebe-bc9a-4c69-a7aa-e0df390a625f","year":2018},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:2384706482cd8eceab714234a1e5ae4978a9aac3f9921a5936546cb6f9a4ec12","observation_id":"df84d74b-44f9-4720-93f7-a92cce382c1c","resolution":{"observed_at":"2026-05-15T12:00:00.401814Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"742c29d6-1d75-4d8b-87f6-bb3cde0d7ea7","year":2024},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:0257c0793cfae89654e393394e2e3a5c8750f65c9e9c979c70f897c5f595ab58","observation_id":"040dcdfc-c036-4e61-8f75-a0c83d71980a","resolution":{"observed_at":"2026-05-15T12:00:00.407103Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ella-v: Stable neural codec language modeling with alignment-guided sequence reordering","venue":null,"work_id":"6fc2cd59-9753-40d1-b41a-63df0512d2f4","year":2024},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:08091ad0b5b14f234523b9d0a928500e460a02d01e4c0d3de55b906178b54e0c","observation_id":"be7bc2df-9edf-4a90-9cfa-a6f0aff4ea13","resolution":{"observed_at":"2026-05-15T12:00:00.413747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Roformer: Enhanced transformer with rotary position embedding.Neurocomputing, 568:127063","venue":null,"work_id":"d1af47bd-dd90-4dff-826b-302fd98831d3","year":null},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:95f35b3418678c345b019c356d73e4180a1585e544e1afec86ec4612f9864cac","observation_id":"6e2656ce-a6d0-493f-b076-b597206a1946","resolution":{"observed_at":"2026-05-15T12:00:00.385031Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"UTMOS: UTokyo-SaruLab System for V oiceMOS Challenge","venue":null,"work_id":"e5f68f57-895e-4a4a-85a0-6958f1dee195","year":null},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:dec1178280c78fa9288c838c28989731b2cdf00d40306efad990cc00da24898e","observation_id":"a6c6e768-8556-4d06-b5e7-30e87bdc5d7a","resolution":{"observed_at":"2026-05-15T12:00:00.395139Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"36855e42-b9cc-4e35-81a3-883c3ff1ac9c","year":2022},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:8b9a64ac649a67c32e4542313c07083992e9a7d83bbd823342f31e0d155e2de1","observation_id":"28390b1a-4208-49ef-befb-ea2cd143fcc5","resolution":{"observed_at":"2026-05-15T12:00:00.367463Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01710","last_updated":"2025-03-03T16:23:10Z","snapshot_observed_at":"2026-07-06T20:45:50.730195Z","submitted_at":"2025-03-03T16:23:10Z","title":"Spark-TTS: An Efficient LLM-Based Text-to-Speech Model with Single-Stream Decoupled Speech Tokens","version":1},"cited_work":{"arxiv_id":"2503.01710","doi":"10.48550/arxiv.2503.01710","metadata_source":"pith","pith_arxiv_id":"2503.01710","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Spark-TTS: An Efficient LLM-Based Text-to-Speech Model with Single-Stream Decoupled Speech Tokens","venue":"cs.SD","work_id":"31f99dad-40ae-4a19-aeff-eafa54f5b42a","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"cited_paper":"/paper/2503.01710","citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:a09eed90ddfb8d6776126fa179ba29ac0a79c49562e1f4df201dcaca5b75c8e3","observation_id":"d5fc9a45-8b02-43b6-939b-541e33fa06e6","resolution":{"observed_at":"2026-05-17T09:39:51.552824Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Skerry-Ryan, Daisy Stanton, Yonghui Wu, Ron J","venue":null,"work_id":"773d9a42-86c1-4b63-8a9b-10c47d551f2f","year":2017},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:3bc5b2c26afb63faa2b561de8c68bab4fa4f7f11eb54127c32b4bf19818bd4d9","observation_id":"def93211-0a50-454b-a80e-72cc5994cdcc","resolution":{"observed_at":"2026-05-15T12:00:00.313086Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"MaskGCT: Zero-shot text- to-speech with masked generative codec transformer","venue":null,"work_id":"acbf400c-d345-4d5e-bda9-7cd27235d34b","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:dd8722bf0cb852c4a1e739f0f9badfd580ba94252398c7c4e0a521bae6705f94","observation_id":"20a97d3e-33e2-4976-84e0-e16c9c3195ea","resolution":{"observed_at":"2026-05-15T12:00:00.325986Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Con- vnext v2: Co-designing and scaling convnets with masked autoencoders","venue":null,"work_id":"33ff4c2c-f690-46cb-8f54-52e96e467062","year":2023},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:9a35f5ee1ad142cd788ffe4ad4f6785f77d57d1821b0ca1806e5bbb03665c228","observation_id":"afa13a78-7126-4280-825c-a4ea4cffc492","resolution":{"observed_at":"2026-05-15T12:00:00.312127Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lipvoicer: Generating speech from silent videos guided by lip reading","venue":null,"work_id":"371e54cc-c77a-4dc0-9ccf-a5a08b718122","year":2024},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:149a8899077f64c34f8c8beec45f2d19f56c91c6c55cf3f2c92d96ff4f0c8717","observation_id":"5d465546-025b-4237-8d89-9df19158282e","resolution":{"observed_at":"2026-05-15T12:00:00.333858Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Simultaneous mod- eling of spectrum, pitch and duration in hmm-based speech synthesis","venue":null,"work_id":"cdcbcd35-5eab-4674-b1a8-9177bc7297f8","year":1999},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:97dff7fec95e9494f5ce0a0775ec9ae02149d63b727185c99b71537fac124039","observation_id":"cd555d55-37e6-4687-9cce-3c7e2055ecb8","resolution":{"observed_at":"2026-05-15T12:00:00.331620Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Weiss, Ye Jia, Zhifeng Chen, and Yonghui Wu","venue":null,"work_id":"fada70e0-1192-4c44-8149-ca183806660b","year":2019},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:d40d2a96b173781e4374c4ad4e4eb0e7a2ed9d92efbe6bf6a922674dadc52e9a","observation_id":"63454965-20ff-4690-b961-e1a8fdbdebb8","resolution":{"observed_at":"2026-05-15T12:00:00.297574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.03926","last_updated":"2023-03-07T14:31:55Z","snapshot_observed_at":"2026-07-06T14:59:43.014800Z","submitted_at":"2023-03-07T14:31:55Z","title":"Speak Foreign Languages with Your Own Voice: Cross-Lingual Neural Codec Language Modeling","version":1},"cited_work":{"arxiv_id":"2303.03926","doi":null,"metadata_source":"pith","pith_arxiv_id":"2303.03926","snapshot_observed_at":"2026-07-08T00:04:22.393626Z","title":"Speak foreign languages with your own voice: Cross-lingual neural codec language modeling","venue":"cs.CL","work_id":"0825378e-00d8-4678-bea7-360d1adc8eef","year":2023},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"cited_paper":"/paper/2303.03926","citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:d3babba90f64a03dbfd5e796a45e1bf2645ce2b7d04993999ab40c13e8f759b5","observation_id":"84b0e2ae-172b-42a0-a79f-bcfdf1f1c229","resolution":{"observed_at":"2026-05-15T11:59:59.597146Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"From speaker to dubber: Movie dubbing with prosody and duration consistency learning","venue":null,"work_id":"6591e7c2-82a7-47d9-b4aa-c458178ec5de","year":2024},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:cc700e7caa6d02fb2ab4baa8372b7b188f5743a9b03780b9376db57e94b8781b","observation_id":"4d7cd245-f93b-4952-a2d6-0d65447059cd","resolution":{"observed_at":"2026-05-15T12:00:00.300906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Prosody- enhanced acoustic pre-training and acoustic-disentangled prosody adapting for movie dubbing","venue":null,"work_id":"59af5db3-96da-416c-b698-65c01ab7f952","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:02b443dd7f6781ac31070a04a7f7a4580aa845a66ecb398a5661224e583b2b35","observation_id":"0899a3c2-894a-44ea-8591-f54b72650c29","resolution":{"observed_at":"2026-05-15T12:00:00.323140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchro- nization","venue":null,"work_id":"5f3eca02-6e63-4217-a11c-5b797fad19d2","year":2025},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:dadfd8e814ed9921e6f91e090f67de905e7fc4252844d4994d6123c5ee91d2c0","observation_id":"271bd3ac-dcb3-4798-ad9f-81fdbb636d86","resolution":{"observed_at":"2026-05-15T12:00:00.355602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5ebee16b-2d80-4df2-80a6-183527c0ebdd","year":null},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:d99022b074043595fd1d61b371979ab373a70040f38b46071f81a9325894ae34","observation_id":"648d5ba8-dafb-47f4-ab5b-a6358189169e","resolution":{"observed_at":"2026-05-15T12:00:00.291487Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The light fusion network is implemented as a linear projection layer","venue":null,"work_id":"da262879-e762-4495-9c6f-8b15231039fe","year":null},"citing_paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","version":4},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-15T11:56:13.914121Z"},"links":{"citing_paper":"/paper/2603.14267"},"observation_digest":"sha256:25093bf0c25f326578bdd0d6d56141c05c9d28767da05dfd8899ba7868d2c607","observation_id":"85d3807e-bfaa-495c-8b10-c15935c3ce5c","resolution":{"observed_at":"2026-05-15T12:00:00.359070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2603.14267","last_updated":"2026-04-03T07:45:32Z","latest_version":4,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T22:49:06.164294Z","submitted_at":"2026-03-15T07:53:23Z","title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization"},"reference_resolution":{"displayed":66,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":5,"verified_exact":5,"verified_fuzzy":56},"total_outbound_references":66},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 66 of 66 outbound references and 1 inbound Pith citation observation for arXiv:2603.14267."}