{"as_of":"2026-08-08T09:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e21257dd2858b0ed5bc0b679919d10a67705b23513c107b0d7f09bd1f6bde898","coverage":[{"denominator":24,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:02:22.477354Z","state":"measured"},{"denominator":25,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":25,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:02:20.273561Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T12:02:22.672497Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"cited_work":{"arxiv_id":"2506.00740","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.00740","snapshot_observed_at":"2026-08-07T12:02:22.672497Z","title":"Length Aware Speech Translation for Video Dubbing","venue":"cs.CL","work_id":"d395e436-21fc-4fe0-a2c6-00fa9fc1a4ee","year":2025},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:20.273561Z"},"links":{"cited_paper":"/paper/2506.00740","citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:bc4d9e38eab07fd9edf8c93d700a9c72b9d11a988b45499812685d0198cd7663","observation_id":"50be5103-5baf-4227-a80e-2b28d4433ed9","resolution":{"observed_at":"2026-08-07T12:02:22.724435Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.00740/citation-record","integrity":"/paper/2506.00740/integrity","json":"/paper/2506.00740/citation-record.json","paper":"/paper/2506.00740"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"cited_work":{"arxiv_id":"2506.00740","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.00740","snapshot_observed_at":"2026-08-07T12:02:22.672497Z","title":"Length Aware Speech Translation for Video Dubbing","venue":"cs.CL","work_id":"d395e436-21fc-4fe0-a2c6-00fa9fc1a4ee","year":2025},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:20.273561Z"},"links":{"cited_paper":"/paper/2506.00740","citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:bc4d9e38eab07fd9edf8c93d700a9c72b9d11a988b45499812685d0198cd7663","observation_id":"50be5103-5baf-4227-a80e-2b28d4433ed9","resolution":{"observed_at":"2026-08-07T12:02:22.724435Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:26.713172Z","title":"The duration of translated audio is influ- enced by: (a) the length of the translated text, and (b) the dura- tion model within the text-to-speech (TTS) system","venue":null,"work_id":"dc81d4ea-2279-458e-9732-56e992a78f7f","year":null},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:20.320178Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:72912f1490bcc035d0f50fb874f7e29764813f29b5294079c05a0995a292f34e","observation_id":"d6cd933b-2236-41c7-b09a-e344d07b06c1","resolution":{"observed_at":"2026-08-07T12:02:26.774957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:26.434166Z","title":"Model and Data The ST model used in our experiments is multilingual and jointly trained on Spanish (ES) and Korean (KO) data","venue":null,"work_id":"e2d8dcea-7d4e-445d-94b0-0dadae6e56ce","year":null},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:20.493478Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:091f9dede6066258507eddbc8dc091f80f3b6e49aa94fa03d329aa62b94b620d","observation_id":"55c0d34f-7e87-4ae5-8f9e-480ebab415d9","resolution":{"observed_at":"2026-08-07T12:02:26.482554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:26.548432Z","title":null,"venue":null,"work_id":"e61e9750-3eaf-4557-a4d4-900a44e9243e","year":null},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:20.403503Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:26f066d5be91ed1214af479d30b0cc95ab91eb1acc94a10e9a6145f031877edc","observation_id":"9336c538-4a13-4902-a877-f02753dfcf8f","resolution":{"observed_at":"2026-08-07T12:02:26.610354Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:26.270470Z","title":null,"venue":null,"work_id":"d96d59c6-948f-438d-b4fc-702b5bf19587","year":null},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:20.581943Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:f69c14044d63582bb9cd23d05fffdffcc1de2012cd78270814a1ce1c73fdc1c0","observation_id":"e5019cc5-db24-48f8-85f7-84aae49d17e8","resolution":{"observed_at":"2026-08-07T12:02:26.355715Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:26.067165Z","title":"Our approach leverages predefined length control tokens to generate transla- tions of varying lengths—short, normal, and long—while main- taining high translation quality","venue":null,"work_id":"c4cf2b43-6ab4-4fc1-b875-11dc417fc368","year":null},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:20.678246Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:c46139548506fdb9ac1fce655d3643c4e1f35fb2184453d7ad4e803f624d660b","observation_id":"2af8228e-6317-4b13-8f9d-f1e6e8ee17af","resolution":{"observed_at":"2026-08-07T12:02:26.158753Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:25.899542Z","title":"Leveraging weakly supervised data to improve end-to-end speech-to-text translation,","venue":null,"work_id":"54dcca3f-b983-4005-9425-3632c6520270","year":2019},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:20.739344Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:c311f8f9c28593f3c2d6433e24569d07e0a2994748f1a31146f4d4ab27cd9b62","observation_id":"d44a2f01-7e25-4985-b926-9ae24a81c927","resolution":{"observed_at":"2026-08-07T12:02:25.962999Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:25.756138Z","title":"Large-scale stream- ing end-to-end speech translation with neural transducers,","venue":null,"work_id":"1bac9c86-12c2-4b54-88f7-09e89157425e","year":2022},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:20.844471Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:10aa4aefd782bf3e2d14a111cb83b58f7d67eda37be3d90d14b14170d4871d1c","observation_id":"e4c0e5fa-9e0c-4500-8491-37fb26954bee","resolution":{"observed_at":"2026-08-07T12:02:25.805883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:25.606258Z","title":"Revisiting end-to-end speech-to-text translation from scratch,","venue":null,"work_id":"8bd872b4-6259-4784-816a-ab68d36ad707","year":2022},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:20.972785Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:2aea7bf52dcecb01cf3ea72c7c81ebd0e8b57f65a5e8f7d205f61d0685fa0b23","observation_id":"5f522bd4-46cd-490f-911b-a374363298d9","resolution":{"observed_at":"2026-08-07T12:02:25.677659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:25.415921Z","title":"Videodubber: machine translation with speech-aware length control for video dubbing,","venue":null,"work_id":"542e1c42-a73b-4d98-8c11-ea8b02fe0dd2","year":2023},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:21.090318Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:bb7a6305e9827fcbca491e6a103228e68c2054fbc2d4b677190cd11984dd758e","observation_id":"80d6cb22-325f-485a-a43f-4405985b0eea","resolution":{"observed_at":"2026-08-07T12:02:25.469350Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:25.229903Z","title":"Controlling machine translation for multiple attributes with additive interven- tions,","venue":null,"work_id":"a4a2535e-9283-4f4b-afcf-9dfaa177c746","year":2021},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:21.144276Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:38c5e34b06aed67009968eea4aa5ec6721a68f4aed897f580f226e7841f8d4ed","observation_id":"6f8690b8-3172-4649-860e-18a7fc8f9f60","resolution":{"observed_at":"2026-08-07T12:02:25.307976Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:25.065565Z","title":"Is 42 the answer to ev- erything in subtitling-oriented speech translation?","venue":null,"work_id":"ff53f1a4-5e6e-44bf-8c61-7c9104a4e5a8","year":2020},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:21.231253Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:eb412125f05f81869bd02ceeedb362498b1ddcf6e3a2c13e3a32f7d65b473356","observation_id":"c891b55d-409f-43c4-9d31-54d915e32e42","resolution":{"observed_at":"2026-08-07T12:02:25.122444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:24.891039Z","title":"HW-TSC’s participa- tion in the IWSLT 2022 isometric spoken language translation,","venue":null,"work_id":"6c7c1102-ca19-4b37-88e1-7544f07ecd02","year":2022},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:21.321075Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:5410682c5eb8e149d107dc5fc458476da284317628e0ac766af55d35768f2fc2","observation_id":"77e30b5e-5581-4f16-8001-2021b968ed32","resolution":{"observed_at":"2026-08-07T12:02:24.977450Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:24.718596Z","title":"Adapting end-to-end speech recognition for readable subtitles,","venue":null,"work_id":"4ea5b56f-506c-4b7a-982d-b01541025d29","year":2020},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:21.444140Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:b94aa02554aa2bd0b234bc96b0c610ad105320d2a8bbef382345f00b8739899b","observation_id":"786a7a35-b98e-4a3d-b333-e4cdf9587e0e","resolution":{"observed_at":"2026-08-07T12:02:24.782957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:24.413848Z","title":"Length- aware NMT and adaptive duration for automatic dubbing,","venue":null,"work_id":"7a688db4-4213-451c-911b-0b0e82b59907","year":2023},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:21.522466Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:5675bda7477b824906b84a3deb480410435496c1006cbe16c896fc22be83cd10","observation_id":"94eab385-7f58-4ef7-a5a4-0a720508dfa6","resolution":{"observed_at":"2026-08-07T12:02:24.579489Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:24.135002Z","title":"Machine translation verbosity control for automatic dubbing,","venue":null,"work_id":"de39bac7-7856-4620-9940-eade4379f27c","year":2021},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:21.641051Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:119bf5a4ec3be9be5213b2383df38cbc290a92056a2fc2074be157d8e4ec928a","observation_id":"63532acc-da9b-4a64-8027-a88178be48d5","resolution":{"observed_at":"2026-08-07T12:02:24.232143Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:24.012005Z","title":"Isochrony-aware neural machine translation for automatic dub- bing,","venue":null,"work_id":"c3bd7cbf-6888-43b1-b52e-c2573580c7a6","year":2022},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:21.720649Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:3cc50bec61fb7c23217ee09845a6bbd9cce87316716d5850c9be76eb9a66c627","observation_id":"04a7a3fd-aef3-43e6-bca2-b6d024091488","resolution":{"observed_at":"2026-08-07T12:02:24.056077Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:23.856924Z","title":"Isometric neural machine translation us- ing phoneme count ratio reward-based reinforcement learning,","venue":null,"work_id":"9f977ccb-f09a-4802-8a15-48c1cbb37203","year":2024},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:21.815996Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:eae01bb7a87fb2dd1cb5eb17040bfbb3f264a815110ee1ef62716ae6047a99ac","observation_id":"1e40c525-7a80-446e-9302-88fa9f79df3c","resolution":{"observed_at":"2026-08-07T12:02:23.923490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:23.641681Z","title":"FastSpeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":"9e25215c-5771-4a67-89e6-77195b6da249","year":2021},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:21.932443Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:e411f293268f18440973be0a9b1b97dc7aff91fb234e135cbf2df5eb5cc5ee1e","observation_id":"c164e01c-07f4-4b11-ab56-17cadb7bef10","resolution":{"observed_at":"2026-08-07T12:02:23.764767Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:23.491235Z","title":"Conformer: Convolution-augmented transformer for speech recognition,","venue":null,"work_id":"618a4921-f544-44af-8905-d63f09752e6a","year":2020},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:22.031806Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:5ec3d4d3bd2a42838a4d771a9bcb7b4a1588daab40fb370ed3d0db25d991bf3a","observation_id":"16f9d6ea-13d9-43c2-a79f-a6925629b3d5","resolution":{"observed_at":"2026-08-07T12:02:23.548438Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:23.342878Z","title":"Hybrid CTC/attention architecture for end-to-end speech recog- nition,","venue":null,"work_id":"154a5e4c-db78-49a8-a497-2e22d41362b6","year":2017},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:22.090541Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:50404342c455f038856e1669049ff28ba2f5cfd0e8a99c7d56906a5fbba70e81","observation_id":"1c399852-892d-4cbc-b3d9-4033bca30e9c","resolution":{"observed_at":"2026-08-07T12:02:23.409250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:23.102422Z","title":"Fleurs: Few-shot learning evaluation of universal representations of speech,","venue":null,"work_id":"172b3ac2-4ecd-490b-876e-fdc19cde2bfb","year":2022},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:22.186666Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:138f9c17be62a96ebd21f6d94c1709b9b9f613d9309647c9fcc60d5f1e7b69bf","observation_id":"0b7d8599-7d6a-4c05-a552-fd47b114cc47","resolution":{"observed_at":"2026-08-07T12:02:23.206909Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:22.937820Z","title":"LeanSpeech: The Microsoft lightweight speech synthesis system for limmits challenge 2023,","venue":null,"work_id":"28381dff-1f38-427f-af7c-bbad5078f9cc","year":2023},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:22.337542Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:1833d9d33327cfa4aa74e28953d739c3c3c8cbae02d2ed6de71abc98e7ad8b12","observation_id":"b412060a-49bc-47aa-a46f-c407124203a7","resolution":{"observed_at":"2026-08-07T12:02:23.010430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:02:22.772858Z","title":"A call for clarity in reporting BLEU scores,","venue":null,"work_id":"3d454081-8963-40d8-accf-478dccf5f9b5","year":2018},"citing_paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:22.477354Z"},"links":{"citing_paper":"/paper/2506.00740"},"observation_digest":"sha256:d6b60e1d3fcc8b28924964209c136c888453255f345a0f8d85e2e4c7a6861823","observation_id":"930721e1-96fd-4a3e-a177-4d0affc1bfeb","resolution":{"observed_at":"2026-08-07T12:02:22.869228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.00740","last_updated":"2025-05-31T23:01:50Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T11:56:44.975449Z","submitted_at":"2025-05-31T23:01:50Z","title":"Length Aware Speech Translation for Video Dubbing"},"reference_resolution":{"displayed":24,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":2,"verified_exact":1,"verified_fuzzy":21},"total_outbound_references":24},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 24 of 24 outbound references and 1 inbound Pith citation observation for arXiv:2506.00740."}