{"as_of":"2026-08-20T04:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e8827efd8bf770867875af7b79f5782500351aa2fc9e7b971de20da12d98012d","coverage":[{"denominator":32,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":32,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T19:16:37.892896Z","state":"measured"},{"denominator":34,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":34,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:15:59.897627Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-16T21:51:17.764632Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.00377","snapshot_observed_at":"2026-08-07T11:15:59.897627Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02995","last_updated":"2025-06-03T15:29:52Z","snapshot_observed_at":"2026-08-16T20:21:18.945975Z","submitted_at":"2025-06-03T15:29:52Z","title":"It's Not a Walk in the Park! Challenges of Idiom Translation in Speech-to-text Systems","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T11:15:59.897627Z"},"links":{"cited_paper":"/paper/2502.00377","citing_paper":"/paper/2506.02995"},"observation_digest":"sha256:8838d5aede84322b632b196789d8863684aa4b371f2ada8d205f439bf77ed7ea","observation_id":"0db557a4-ee39-43a5-af0a-938b1132a85a","resolution":{"observed_at":"2026-08-07T11:15:59.897627Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"cited_work":{"arxiv_id":"2502.00377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.00377","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a2294b94-d6ed-481c-b90f-1ad6868c8654","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-15T09:09:05.619478Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2502.00377","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:e483f627a543450eb127c5027b0ff114401937803630f82e8f308166aea0ade4","observation_id":"3f40e10f-5ebd-426b-94cd-250e0379a231","resolution":{"observed_at":"2026-05-16T21:51:17.766208Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2502.00377/citation-record","integrity":"/paper/2502.00377/integrity","json":"/paper/2502.00377/citation-record.json","paper":"/paper/2502.00377"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1612.01744","last_updated":"2016-12-06T10:48:56Z","snapshot_observed_at":"2026-08-14T21:26:55.522063Z","submitted_at":"2016-12-06T10:48:56Z","title":"Listen and Translate: A Proof of Concept for End-to-End Speech-to-Text Translation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.01744","snapshot_observed_at":"2026-08-09T19:16:37.771901Z","title":"Listen and translate: A proof of concept for end-to-end speech-to-text translation,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.771901Z"},"links":{"cited_paper":"/paper/1612.01744","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:805f828c00598bdf94a2b85534688c43b4f0fc4361b33eea8097b6b3e7023054","observation_id":"71a4c16e-6d47-4d59-83f2-85c970d3b612","resolution":{"observed_at":"2026-08-09T19:16:37.771901Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.02967","last_updated":"2022-09-13T17:00:55Z","snapshot_observed_at":"2026-08-16T17:08:48.467808Z","submitted_at":"2022-04-06T17:59:22Z","title":"Enhanced Direct Speech-to-Speech Translation Using Self-supervised Pre-training and Data Augmentation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.02967","snapshot_observed_at":"2026-08-09T19:16:37.776878Z","title":"Enhanced direct speech-to-speech translation using self-supervised pre-training and data augmentation,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.776878Z"},"links":{"cited_paper":"/paper/2204.02967","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:9e4729ebb77c462814117978f0bcf8ad05928ece43a02b8d4600d201833fe6a7","observation_id":"6c869978-87ae-4c77-b825-c11b644a453c","resolution":{"observed_at":"2026-08-09T19:16:37.776878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:38.271286Z","title":"Leveraging weakly supervised data to improve end-to-end speech-to-text translation,","venue":null,"work_id":"8fe279b2-0412-4eb5-949c-7183c7ce891b","year":2019},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.781658Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:cfee0b971724593dabb470ed6989e37824b0828b7fcbecdc35a4ff4fac1328b7","observation_id":"73ac5f6b-8f20-45ee-a374-999f7122000d","resolution":{"observed_at":"2026-08-09T19:16:38.275202Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:38.259365Z","title":"Analyzing asr pretraining for low-resource speech-to-text translation,","venue":null,"work_id":"1c3e3923-9882-40bb-a431-77701159b8f4","year":2020},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.785901Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:66583df0894281d700af556cc29204f4dec2804d8c1d661e2a196d0c36bb45e5","observation_id":"5b07f8f3-af5d-41cc-abea-f4bab83b0fed","resolution":{"observed_at":"2026-08-09T19:16:38.263304Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.02490","last_updated":"2020-10-13T05:25:01Z","snapshot_observed_at":"2026-08-12T12:52:55.791519Z","submitted_at":"2020-06-03T19:28:36Z","title":"Self-Training for End-to-End Speech Translation","version":2},"cited_work":{"arxiv_id":"2006.02490","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.02490","snapshot_observed_at":"2026-08-09T19:16:38.074927Z","title":"Self-Training for End-to-End Speech Translation","venue":"cs.CL","work_id":"522373da-496e-4a67-b4b9-96aafde7e1b1","year":2020},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.789794Z"},"links":{"cited_paper":"/paper/2006.02490","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:538b4eccf2bc7dabbed6f4a6b961cfa23168a4edfb36fc9ddd70f16d868ea7df","observation_id":"daa731d3-159c-40d9-b1a7-f4451858b50b","resolution":{"observed_at":"2026-08-09T19:16:38.078778Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.06678","last_updated":"2021-04-14T07:44:52Z","snapshot_observed_at":"2026-08-16T18:31:44.148344Z","submitted_at":"2021-04-14T07:44:52Z","title":"Large-Scale Self- and Semi-Supervised Learning for Speech Translation","version":1},"cited_work":{"arxiv_id":"2104.06678","doi":null,"metadata_source":"pith","pith_arxiv_id":"2104.06678","snapshot_observed_at":"2026-08-09T19:16:38.060114Z","title":"Large-Scale Self- and Semi-Supervised Learning for Speech Translation","venue":"cs.CL","work_id":"97642d4a-839f-42f1-afb9-f3701699d251","year":2021},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.794075Z"},"links":{"cited_paper":"/paper/2104.06678","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:cef555726790871eaf532f512ad93eef0197cbe82f77af1188aa55cd77b7ee99","observation_id":"16240c3b-62e8-424d-9fe3-98139f24601f","resolution":{"observed_at":"2026-08-09T19:16:38.064399Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:38.248684Z","title":"On the integration of speech recognition and statistical machine translation,","venue":null,"work_id":"35e73433-2935-4106-8644-7f2ecbc723f0","year":2005},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.798436Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:714f8d0c1afbfc46a64b10ba8a64ce272339937888bc07e1642093609786c7a5","observation_id":"3786c8b8-75f5-4085-8691-fc2d004b7c65","resolution":{"observed_at":"2026-08-09T19:16:38.252457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:38.237419Z","title":"Integrated n-best re-ranking for spoken language translation","venue":null,"work_id":"155299d0-a395-4818-a8aa-b07f2d2ba618","year":2005},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.802110Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:2cf73e74d4458479244a3d323d43cbf2dfda90b4512b6ad4ab69c1a63e6312bb","observation_id":"93b24755-3919-408c-8fda-56b78a60279c","resolution":{"observed_at":"2026-08-09T19:16:38.241695Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:38.225586Z","title":"Neural lattice search for speech recognition,","venue":null,"work_id":"bfa0e869-7af0-44ed-a0fc-8eb1e5e96965","year":2020},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.805854Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:7818e1dae0f3a52f4dbfc9c909bff836390a24e3b9fc9187c174fc617b8af3d0","observation_id":"12c6dc73-2e67-4b66-b77c-3a881a428dbb","resolution":{"observed_at":"2026-08-09T19:16:38.229768Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:38.214298Z","title":"A new decoder for spoken language trans- lation based on confusion networks,","venue":null,"work_id":"15af44ef-12f1-4524-9d85-b867d3c2f4f3","year":2005},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.809405Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:6b9cb203cd89ba07ce59b967f22344194135cff2c9c126b2c2c4c348089eb169","observation_id":"f11eac98-c439-4b7b-ac42-470edb234e38","resolution":{"observed_at":"2026-08-09T19:16:38.218253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:38.201974Z","title":"Neural speech translation using lattice transformations and graph networks,","venue":null,"work_id":"6cfe451d-b9a1-4a5c-9f57-fd2c52dd05e0","year":2019},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.813484Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:ae255d8062aacf825f2bc5e0489486db16c35cdfa5504d5f2627aa2aa928f2a1","observation_id":"ac714583-4672-4a1f-8f50-5b6f5e127649","resolution":{"observed_at":"2026-08-09T19:16:38.206776Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1906.01617","last_updated":"2019-06-04T17:51:03Z","snapshot_observed_at":"2026-08-15T10:11:00.822080Z","submitted_at":"2019-06-04T17:51:03Z","title":"Self-Attentional Models for Lattice Inputs","version":1},"cited_work":{"arxiv_id":"1906.01617","doi":null,"metadata_source":"pith","pith_arxiv_id":"1906.01617","snapshot_observed_at":"2026-08-09T19:16:38.044458Z","title":"Self-Attentional Models for Lattice Inputs","venue":"cs.CL","work_id":"9dc9dc82-7714-45ee-9619-6eb54fce16be","year":2019},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.817351Z"},"links":{"cited_paper":"/paper/1906.01617","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:fe46b9e1845b93749ab945a7c607429736654d6b40b3b252c1205c7a7393b84c","observation_id":"781bcee2-c4d9-4ad0-a27e-a23ca24e31f0","resolution":{"observed_at":"2026-08-09T19:16:38.048825Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:38.190324Z","title":"Spoken language translation using automatically transcribed text in training,","venue":null,"work_id":"807b9d23-6f20-4c21-ba11-d66b3c8342da","year":2012},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.821291Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:29332aad02900b9870bc40c5fc095413234c44c3b239a9a6ad8f79cd841ee775","observation_id":"d2c0d79c-189d-432b-a554-4febaf08f1ed","resolution":{"observed_at":"2026-08-09T19:16:38.194270Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11430","last_updated":"2019-10-28T11:16:53Z","snapshot_observed_at":"2026-08-18T08:50:45.395056Z","submitted_at":"2019-09-25T12:13:43Z","title":"Breaking the Data Barrier: Towards Robust Speech Translation via Adversarial Stability Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.11430","snapshot_observed_at":"2026-08-09T19:16:37.825059Z","title":"Breaking the data barrier: Towards robust speech translation via adversarial stability training,","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.825059Z"},"links":{"cited_paper":"/paper/1909.11430","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:c64af67425f8ee56aa53313a51548635262d8bfb55fe625b669562b0fd563d78","observation_id":"1f39ced0-f968-4657-9838-e4139964e220","resolution":{"observed_at":"2026-08-09T19:16:37.825059Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.10238","last_updated":"2019-10-22T21:24:24Z","snapshot_observed_at":"2026-08-19T10:42:51.250397Z","submitted_at":"2019-10-22T21:24:24Z","title":"Robust Neural Machine Translation for Clean and Noisy Speech Transcripts","version":1},"cited_work":{"arxiv_id":"1910.10238","doi":null,"metadata_source":"pith","pith_arxiv_id":"1910.10238","snapshot_observed_at":"2026-08-09T19:16:38.018321Z","title":"Robust Neural Machine Translation for Clean and Noisy Speech Transcripts","venue":"cs.CL","work_id":"198e1ac6-8a29-4866-ac62-440c40ebb793","year":2019},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.828855Z"},"links":{"cited_paper":"/paper/1910.10238","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:af63a38203f3555f2aedcaba1c90deb2ecad85647b480ed1e3a9ad3b88ef742b","observation_id":"28afe047-e2aa-42cd-a049-22c8607830a9","resolution":{"observed_at":"2026-08-09T19:16:38.022516Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.00573","last_updated":"2021-05-02T23:22:49Z","snapshot_observed_at":"2026-08-16T18:27:32.962311Z","submitted_at":"2021-05-02T23:22:49Z","title":"Searchable Hidden Intermediates for End-to-End Models of Decomposable Sequence Tasks","version":1},"cited_work":{"arxiv_id":"2105.00573","doi":null,"metadata_source":"pith","pith_arxiv_id":"2105.00573","snapshot_observed_at":"2026-08-09T19:16:38.000400Z","title":"Searchable Hidden Intermediates for End-to-End Models of Decomposable Sequence Tasks","venue":"cs.CL","work_id":"de3fc8f9-edb6-4791-bca1-51702da2a2b6","year":2021},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.832729Z"},"links":{"cited_paper":"/paper/2105.00573","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:a26b92855b92d00757ed2bc203ee743988d5ea2ad582afdb30c05ac35e6d85ca","observation_id":"e8f5904d-8cad-4feb-9970-f67b51a6c443","resolution":{"observed_at":"2026-08-09T19:16:38.006903Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:38.178844Z","title":"Fast-md: Fast multi-decoder end-to-end speech translation with non-autoregressive hidden intermediates,","venue":null,"work_id":"fe46eac8-4692-4169-b814-aae534fd9348","year":2021},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.836438Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:216969689abaaae8cf3df29d4341ed223e621d79cd0819f13ece3ada9d344153","observation_id":"63be6bef-3aa4-4cb9-a139-eecf36350405","resolution":{"observed_at":"2026-08-09T19:16:38.183099Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07258","last_updated":"2022-07-12T23:45:14Z","snapshot_observed_at":"2026-08-02T09:20:40.804790Z","submitted_at":"2021-08-16T17:50:08Z","title":"On the Opportunities and Risks of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07258","snapshot_observed_at":"2026-08-09T19:16:37.839958Z","title":"On the opportunities and risks of foundation models,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.839958Z"},"links":{"cited_paper":"/paper/2108.07258","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:7a14b499cb33cb828fb53a06ac34301b13cf4e7b933245d151adcd6567ddc9c7","observation_id":"9ca7d3c5-2e50-42e4-8f3c-37a13f6c4f4f","resolution":{"observed_at":"2026-08-09T19:16:37.839958Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:37.843562Z","title":"Pre-trained models: Past, present and future,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.843562Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:c4ccbf22ed97f6b9a069c0073538124b7ff9918b8d4c5f59087edc9b9bee7dab","observation_id":"8ed147d5-4938-4c3f-9ed0-e93c9a35ff3f","resolution":{"observed_at":"2026-08-09T19:16:37.843562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2102.01547","last_updated":"2021-12-29T10:10:52Z","snapshot_observed_at":"2026-08-16T18:48:16.486137Z","submitted_at":"2021-02-02T15:19:41Z","title":"WeNet: Production oriented Streaming and Non-streaming End-to-End Speech Recognition Toolkit","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2102.01547","snapshot_observed_at":"2026-08-09T19:16:37.847173Z","title":"Wenet: Production oriented streaming and non-streaming end-to-end speech recognition toolkit,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.847173Z"},"links":{"cited_paper":"/paper/2102.01547","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:7dbd38fdd4f7cdf7a40f9a4384d0e6e4444e2718cd34b85d7a97c7fb116d0cc8","observation_id":"d4cffacc-9597-448f-8c66-fdefb9c92978","resolution":{"observed_at":"2026-08-09T19:16:37.847173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.03939","last_updated":"2023-06-06T12:48:48Z","snapshot_observed_at":"2026-08-19T05:22:36.233314Z","submitted_at":"2022-04-08T08:59:33Z","title":"GigaST: A 10,000-hour Pseudo Speech Translation Corpus","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.03939","snapshot_observed_at":"2026-08-09T19:16:37.850653Z","title":"Gi- gast: A 10,000-hour pseudo speech translation corpus,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.850653Z"},"links":{"cited_paper":"/paper/2204.03939","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:9937a072382fa71126305399500e6f2ff2da9bbffae476a205ede27d016521dc","observation_id":"62c0c51c-d45b-4131-a432-e44537495cef","resolution":{"observed_at":"2026-08-09T19:16:37.850653Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:38.160691Z","title":"Faster Algorithms for Longest Common Substring,","venue":null,"work_id":"71510853-0eef-4517-aaca-c3cc1a416205","year":2021},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.854393Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:8abadaaca9416f74368805540a85f24b0a4fa75d13149aa0ffd07b8d4abb05b4","observation_id":"9bdf3221-04a5-491f-9226-7f8017b1ed4f","resolution":{"observed_at":"2026-08-09T19:16:38.164596Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.05604","last_updated":"2022-03-21T20:00:14Z","snapshot_observed_at":"2026-08-16T18:10:35.546569Z","submitted_at":"2021-07-12T17:40:43Z","title":"Direct speech-to-speech translation with discrete units","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.05604","snapshot_observed_at":"2026-08-09T19:16:37.857796Z","title":"Direct speech-to-speech translation with discrete units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.857796Z"},"links":{"cited_paper":"/paper/2107.05604","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:50155396abbdc93be7a246dc3cb56bfc8bc73ea6efa245e246c02a76a384f04b","observation_id":"af77003e-b8f4-4571-b116-5b776e43fa43","resolution":{"observed_at":"2026-08-09T19:16:37.857796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:37.861564Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.861564Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:a96786c10f7329d8575a90a118b3ea9ed5aa3bf8dec1db8d28d399ae3d3ce62d","observation_id":"9442da8d-d099-405e-b24c-8dbec3965018","resolution":{"observed_at":"2026-08-09T19:16:37.861564Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2105.01051","last_updated":"2021-10-15T22:04:39Z","snapshot_observed_at":"2026-08-16T18:27:20.253405Z","submitted_at":"2021-05-03T17:51:09Z","title":"SUPERB: Speech processing Universal PERformance Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.01051","snapshot_observed_at":"2026-08-09T19:16:37.865512Z","title":"Superb: Speech processing universal performance benchmark,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.865512Z"},"links":{"cited_paper":"/paper/2105.01051","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:619e348c0200e6c01dfa8418080a682a893a656f193a15a8ac68c9277ec2cc5a","observation_id":"36a1a052-2da7-43d5-a94e-3fa582a15f61","resolution":{"observed_at":"2026-08-09T19:16:37.865512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:37.869345Z","title":"On gener- ative spoken language modeling from raw audio,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.869345Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:a91fb8be3129cbb23662e07fabb497e3e4101d46a6f85bab860068f1381d731a","observation_id":"40867aa0-dbf6-4c5f-b4b6-8bab76284ec5","resolution":{"observed_at":"2026-08-09T19:16:37.869345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.00355","last_updated":"2021-07-27T14:27:27Z","snapshot_observed_at":"2026-08-16T18:34:32.910572Z","submitted_at":"2021-04-01T09:20:33Z","title":"Speech Resynthesis from Discrete Disentangled Self-Supervised Representations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.00355","snapshot_observed_at":"2026-08-09T19:16:37.872786Z","title":"Speech resynthesis from discrete disentangled self-supervised representations,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.872786Z"},"links":{"cited_paper":"/paper/2104.00355","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:8ee7aaf3772a001b3c05bf856363349ec82b10e83a0fabc3cb05cec965482a3d","observation_id":"773af0ec-1f29-4a19-8d89-78a328d50aa4","resolution":{"observed_at":"2026-08-09T19:16:37.872786Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:38.135297Z","title":"Speech-to-speech translation between untranscribed unknown languages,","venue":null,"work_id":"d8afbe78-3328-48ef-8ac4-9226454e5d1b","year":2019},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.877863Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:ea44aeccb35f1c80094eb6f59975246bc2d8bf1c907f1c9204f30071d78ed1c7","observation_id":"415b8310-443a-4da4-a44e-d58126dbd1d9","resolution":{"observed_at":"2026-08-09T19:16:38.139469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:38.123442Z","title":"Uwspeech: Speech to speech translation for unwritten languages,","venue":null,"work_id":"74269cfe-4f4f-4cc7-ae45-8e2d830de0f0","year":2021},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.881420Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:ed25c6bb2cb271c16a65704b503d70c8ceb7a74789fdcc261fa3b99f961acd63","observation_id":"00e19668-e83b-41f8-aa38-448104400d91","resolution":{"observed_at":"2026-08-09T19:16:38.127604Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.06909","last_updated":"2021-06-13T04:09:16Z","snapshot_observed_at":"2026-08-16T18:17:33.981607Z","submitted_at":"2021-06-13T04:09:16Z","title":"GigaSpeech: An Evolving, Multi-domain ASR Corpus with 10,000 Hours of Transcribed Audio","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.06909","snapshot_observed_at":"2026-08-09T19:16:37.885007Z","title":"Gigaspeech: An evolving, multi- domain asr corpus with 10,000 hours of transcribed audio,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.885007Z"},"links":{"cited_paper":"/paper/2106.06909","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:c3ce66a122d48352148e7d2f2c00a7cbf26fa9600e175b6eb0fbd7d359aece1c","observation_id":"c68a647c-86b7-4515-9d19-a420d719dd5a","resolution":{"observed_at":"2026-08-09T19:16:37.885007Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00390","last_updated":"2021-07-27T04:04:57Z","snapshot_observed_at":"2026-08-16T18:54:42.307692Z","submitted_at":"2021-01-02T07:24:21Z","title":"VoxPopuli: A Large-Scale Multilingual Speech Corpus for Representation Learning, Semi-Supervised Learning and Interpretation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2101.00390","snapshot_observed_at":"2026-08-09T19:16:37.888931Z","title":"V oxpopuli: A large-scale multilingual speech corpus for representation learning, semi-supervised learning and interpretation,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.888931Z"},"links":{"cited_paper":"/paper/2101.00390","citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:57b2d860d3f57135fba8e1fb57371f9bdbfe1ec7a66d0a5d9b0b38a87ca583a1","observation_id":"365fdb58-83d0-4756-99da-81ddc14db160","resolution":{"observed_at":"2026-08-09T19:16:37.888931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:16:38.111734Z","title":"Multilingual denoising pre-training for neural machine translation,","venue":null,"work_id":"b9e0aa4d-7401-462f-9fde-cb0239585882","year":2020},"citing_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-09T19:16:37.892896Z"},"links":{"citing_paper":"/paper/2502.00377"},"observation_digest":"sha256:a8732171ecab42cf9cd02b150f83f360b1f7149b9a969100d4780f2eae0e15b3","observation_id":"fc7f0ef0-03a1-4c49-adf1-441b037de1a1","resolution":{"observed_at":"2026-08-09T19:16:38.115683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-16T09:56:33.713959Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation"},"reference_resolution":{"displayed":32,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":14,"verified_exact":5,"verified_fuzzy":13},"total_outbound_references":32},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 32 of 32 outbound references and 2 inbound Pith citation observations for arXiv:2502.00377."}