{"as_of":"2026-08-06T13:18:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3bc447911c61378b9ad56a7a2ed16ed561e2b55d9541e877a4047681504aa341","coverage":[{"denominator":30,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-17T03:15:04.685150Z","state":"measured"},{"denominator":30,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":30,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2512.01512/citation-record","integrity":"/paper/2512.01512/integrity","json":"/paper/2512.01512/citation-record.json","paper":"/paper/2512.01512"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Making llms better many-to-many speech-to-text translators with curriculum learning","venue":null,"work_id":"033de22c-77bb-4980-9c50-717079533608","year":2025},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:3bd5c5fafbc93b0253e76cb21e439387b04c2f52a19d9bfe0fbc6b6b93ed7138","observation_id":"02eb6966-55f9-472a-8bc2-9ff2510c7883","resolution":{"observed_at":"2026-05-17T03:23:58.426639Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","venue":null,"work_id":"9d102f85-3762-4c50-8e15-62567bb95050","year":2020},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:8c2cd9742d5d2048e9f60a6178f013db3fa15ccdfc6ab74e08fd12b429827dd8","observation_id":"6cda7760-b188-459e-9242-a1c5c698cefb","resolution":{"observed_at":"2026-05-17T03:23:58.442974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11430","last_updated":"2019-10-28T11:16:53Z","snapshot_observed_at":"2026-07-06T08:24:26.498016Z","submitted_at":"2019-09-25T12:13:43Z","title":"Breaking the Data Barrier: Towards Robust Speech Translation via Adversarial Stability Training","version":3},"cited_work":{"arxiv_id":"1909.11430","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11430","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Breaking the data barrier: Towards robust speech translation via adversarial stability training","venue":null,"work_id":"2a07466e-e9b8-49b3-9c91-b5d347ff876b","year":1909},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/1909.11430","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:05dd6f7a19004aeab2851e679d11b3efd784a93e7165f7460ec1e85ae1080f92","observation_id":"93de562e-c2b2-42c6-be9b-4d12469955ff","resolution":{"observed_at":"2026-05-17T03:18:57.172604Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07919","last_updated":"2023-12-21T10:20:42Z","snapshot_observed_at":"2026-07-06T16:47:12.709738Z","submitted_at":"2023-11-14T05:34:50Z","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","version":2},"cited_work":{"arxiv_id":"2311.07919","doi":"10.48550/arxiv.2311.07919","metadata_source":"pith","pith_arxiv_id":"2311.07919","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","venue":"eess.AS","work_id":"d3f033ac-bfa8-4143-9d0d-51f3f5bd3f0e","year":2023},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/2311.07919","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:60005bbaf725759eca3d88dcf9dd9d2d82c1813f92cc45ea0cc6eae288675e87","observation_id":"5a4f7985-4e58-499b-94de-6d9b360f85cf","resolution":{"observed_at":"2026-05-17T03:18:57.209538Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Speech translation and the end-to-end promise: Taking stock of where we are","venue":null,"work_id":"579cbb25-e72c-422f-9b68-2091b8303cbf","year":2020},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:805d3fa6be54922a75cfa588b35dd59b34851a6b85d8df548da48867d29fc4f6","observation_id":"97bc6fba-0bc9-49d2-af27-ba0ed00d97e4","resolution":{"observed_at":"2026-05-17T03:23:58.445463Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11000","last_updated":"2023-05-19T14:41:16Z","snapshot_observed_at":"2026-07-06T15:29:16.851803Z","submitted_at":"2023-05-18T14:23:25Z","title":"SpeechGPT: Empowering Large Language Models with Intrinsic Cross-Modal Conversational Abilities","version":2},"cited_work":{"arxiv_id":"2305.11000","doi":"10.48550/arxiv.2305.11000","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11000","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Speechgpt: Empowering large language models with intrinsic cross-modal conversational abilities","venue":"arXiv (Cornell University)","work_id":"003168ec-b0d0-4979-830c-7df75e11d84b","year":2023},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/2305.11000","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:16a0b8fa82ee514148b2460f8346aca1214fa03a624ffcbfaea649431c228d01","observation_id":"35d34721-1096-4ca2-9fac-63a653d4ea36","resolution":{"observed_at":"2026-05-17T03:18:57.162741Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10759","last_updated":"2024-07-15T14:38:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T14:38:09Z","title":"Qwen2-Audio Technical Report","version":1},"cited_work":{"arxiv_id":"2407.10759","doi":"10.48550/arxiv.2407.10759","metadata_source":"pith","pith_arxiv_id":"2407.10759","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2-Audio Technical Report","venue":"eess.AS","work_id":"c249e63c-cf40-408f-a4ff-fdf68e8cbeb8","year":2024},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/2407.10759","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:cea189fc6474890e3fd5ba91aa3fd95d72d534ecb78d1f8606f36656fae609a7","observation_id":"9a057347-1178-4337-a83c-b1bdb57358cd","resolution":{"observed_at":"2026-05-17T03:18:57.179508Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.10310","last_updated":"2020-10-24T06:07:01Z","snapshot_observed_at":"2026-08-01T17:44:35.425095Z","submitted_at":"2020-07-20T17:53:35Z","title":"CoVoST 2 and Massively Multilingual Speech-to-Text Translation","version":3},"cited_work":{"arxiv_id":"2007.10310","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.10310","snapshot_observed_at":"2026-07-04T20:30:07.211214Z","title":"Covost 2 and massively multilingual speech-to-text translation","venue":null,"work_id":"6c9bffaf-0bd1-4fa9-ac70-56e01e042f08","year":2007},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/2007.10310","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:fd8f14e4afade1eafc030e745de2af66fd75e0a31f2d8779bc448b2de62aee5c","observation_id":"414cddce-ec70-470b-a04a-7cad7466bf41","resolution":{"observed_at":"2026-05-17T03:18:57.176372Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual instruction tuning","venue":null,"work_id":"174d8d95-078f-4544-aa44-55b2660dc7e4","year":2023},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:3ea7d89e3c1e77635af09b2c5ea602ae6fe655c2c7e85473c43072dd9b8b6623","observation_id":"c1f81a7a-d0c1-4cdd-b04a-c033aca281d1","resolution":{"observed_at":"2026-05-17T03:23:58.410579Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Blip-2: Bootstrapping language- image pre-training with frozen image encoders and large language models","venue":null,"work_id":"f94dcce4-025b-4dee-802d-b9808499a8df","year":2023},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:c1c29eb041cfa49f262751e380d2982ba53dacfc3629873dfdbff333f7d335dc","observation_id":"efc59e1e-696b-424b-9e9b-798b61ff751e","resolution":{"observed_at":"2026-05-17T03:23:58.440432Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-07-06T13:13:37.143547Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:1e6370a69d9d19822264bce8858321c19d72a5165d0934bd2ac0cac2e69fa66b","observation_id":"de6847ad-3c01-4237-aed6-52f5a23364f3","resolution":{"observed_at":"2026-05-17T03:18:57.198237Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Robust speech recognition via large-scale weak super- vision","venue":null,"work_id":"17e650af-0bc0-4632-9343-efa35133ce5a","year":2023},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:e6900c1ac3c444064ff5ed28f47e2d8e196885615059158ed379abfbcd062432","observation_id":"b3c62e40-cfb2-4584-9cb4-a9b1f8dca8c1","resolution":{"observed_at":"2026-05-17T03:23:58.413266Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scaling neural machine translation to 200 languages","venue":null,"work_id":"ff84af66-afae-4cca-9708-171038d3ed53","year":2024},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:1c6b06a40873d6305db42adaaf2816052b1d6fff191dc78f1d34e76060eb2be6","observation_id":"2f57c748-6dc0-4169-8d8d-e01cf7fd0732","resolution":{"observed_at":"2026-05-17T03:23:58.423866Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.05474","last_updated":"2020-10-09T04:07:38Z","snapshot_observed_at":"2026-08-05T15:37:21.275994Z","submitted_at":"2020-06-09T19:34:11Z","title":"Improving Cross-Lingual Transfer Learning for End-to-End Speech Recognition with Speech Translation","version":2},"cited_work":{"arxiv_id":"2006.05474","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2006.05474","snapshot_observed_at":"2026-06-29T11:43:23.618139Z","title":"Improving cross-lingual transfer learn- ing for end-to-end speech recognition with speech translation","venue":null,"work_id":"c9a4e1e1-2aff-442a-bb79-b0c961a6eb30","year":2006},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/2006.05474","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:31da26e0d6a8eee53bbed28c412e551e9b758b2727afcac0a887a129e37b9dde","observation_id":"c3007809-8ac5-4f77-84f5-b8b4cb8834f9","resolution":{"observed_at":"2026-05-17T03:18:57.205761Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Joint speech and text machine translation for up to 100 languages","venue":null,"work_id":"cb77991a-1fd8-4fdf-9976-fea2462b5486","year":2025},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:85dc12e1bbcb32b4603e45d2157d1e862dd83b8856270c05ccc7fd6aef83a64e","observation_id":"a4f8f7d9-1ca8-417a-96db-6caead053c7f","resolution":{"observed_at":"2026-05-17T03:23:58.429530Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Perception, reason, think, and plan: A survey on large multimodal reasoning models","venue":null,"work_id":"dc5313b1-5827-4c4a-b6ee-145133a24ca1","year":2025},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:1aa0238e2aaa8025be8eb52ba229a0f1ce823787afb806775b9ad7f76690ab3d","observation_id":"780f36a5-251a-44c5-8e92-baac27af8a8b","resolution":{"observed_at":"2026-05-17T03:23:58.432252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13289","last_updated":"2024-04-08T06:12:52Z","snapshot_observed_at":"2026-08-02T21:51:36.809095Z","submitted_at":"2023-10-20T05:41:57Z","title":"SALMONN: Towards Generic Hearing Abilities for Large Language Models","version":2},"cited_work":{"arxiv_id":"2310.13289","doi":"10.1016/j.ajhg.2009.06.010","metadata_source":"pith","pith_arxiv_id":"2310.13289","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SALMONN: Towards Generic Hearing Abilities for Large Language Models","venue":"cs.SD","work_id":"b06d6def-ea7a-4cbd-8bb3-73f6a81db238","year":2023},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/2310.13289","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:2d7dc8085eee01ac1a7256f20ab935b25dec7effc06ece8a7f3b1f7fdb5ee687","observation_id":"39c63e5d-f237-4a6c-944f-a3c58f34bd77","resolution":{"observed_at":"2026-05-18T02:29:46.683133Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.13264","last_updated":"2025-07-17T16:17:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-17T16:17:37Z","title":"Voxtral","version":1},"cited_work":{"arxiv_id":"2507.13264","doi":"10.48550/arxiv.2507.13264","metadata_source":"pith","pith_arxiv_id":"2507.13264","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V oxtral","venue":"cs.SD","work_id":"d8af79fd-31af-429f-b7c7-b98cea5e46ed","year":2025},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/2507.13264","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:eb5b92524176da6cea3d5dd7891f8aa620909ef9260bd2ffb3e3382d66068303","observation_id":"0f6ce27a-0879-40da-ba58-40b297daf860","resolution":{"observed_at":"2026-05-17T03:18:57.194337Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.17765","last_updated":"2025-09-22T13:26:24Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-22T13:26:24Z","title":"Qwen3-Omni Technical Report","version":1},"cited_work":{"arxiv_id":"2509.17765","doi":"10.48550/arxiv.2509.17765","metadata_source":"pith","pith_arxiv_id":"2509.17765","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen3-Omni Technical Report","venue":"cs.CL","work_id":"ae43e594-8bab-4471-b6af-92a300f6a048","year":2025},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/2509.17765","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:ea08cdceab3d10ed3969fc8b2503b2a71106640a7fc74b91f7491561ac3eff93","observation_id":"8926774b-5ddd-4373-b095-b3f9c88e42a4","resolution":{"observed_at":"2026-05-17T03:18:57.201620Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Common voice: A massively-multilingual speech corpus","venue":null,"work_id":"aaf972ff-8361-45c8-a96a-ad89ce50d3ff","year":2020},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:e82ca1df046678cc0d24f03b73e863242a894d3ac791d613d8423d30a042e218","observation_id":"94d04a59-3e00-4cdc-9492-4568178728ae","resolution":{"observed_at":"2026-05-17T03:23:58.435162Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02481","last_updated":"2025-02-24T17:24:04Z","snapshot_observed_at":"2026-07-06T20:31:03.484385Z","submitted_at":"2025-02-04T16:57:03Z","title":"Multilingual Machine Translation with Open Large Language Models at Practical Scale: An Empirical Study","version":4},"cited_work":{"arxiv_id":"2502.02481","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.02481","snapshot_observed_at":"2026-07-02T08:36:48.608960Z","title":"Multilingual machine translation with open large language models at practical scale: An empirical study","venue":null,"work_id":"961c1e22-e42f-4baa-b60f-3744349e7e8f","year":2025},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/2502.02481","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:f9617c41e816db7bb914a746e9415e0dcb5ff3ad889ad2bc0a5fdfb532f1524f","observation_id":"d02f400c-2141-4d26-bb59-917b93620d99","resolution":{"observed_at":"2026-05-17T03:18:57.190010Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19786","last_updated":"2025-03-25T15:52:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-25T15:52:34Z","title":"Gemma 3 Technical Report","version":1},"cited_work":{"arxiv_id":"2503.19786","doi":"10.1007/978-3-540-48085-3_36","metadata_source":"pith","pith_arxiv_id":"2503.19786","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemma 3 Technical Report","venue":"cs.CL","work_id":"f93e08bf-9e96-409b-8ac6-b8385fd17fd7","year":2025},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/2503.19786","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:3e974d0fd1bfeb6062d482d91677e2417032fdfdf15c03afca4e4842a7135a60","observation_id":"7c79003e-7e08-4af5-b83a-67d9425b7bc7","resolution":{"observed_at":"2026-05-17T03:18:57.165618Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":"2106.09685","doi":"10.4088/pcc.v03n0609","metadata_source":"pith","pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","venue":"cs.CL","work_id":"0426219a-789e-4964-adc8-a04538510818","year":2021},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:848240719db806e06aeaa218ee8c4f0f523dcbc957981649ac1084eeb67f7932","observation_id":"07f4b177-e71c-4ef1-9645-21e064b817b6","resolution":{"observed_at":"2026-05-17T03:18:57.186084Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11596","last_updated":"2023-10-25T03:52:07Z","snapshot_observed_at":"2026-07-06T16:09:09.018523Z","submitted_at":"2023-08-22T17:44:18Z","title":"SeamlessM4T: Massively Multilingual & Multimodal Machine Translation","version":3},"cited_work":{"arxiv_id":"2308.11596","doi":"10.48550/arxiv.2308.11596","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11596","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"C.; Dale, D.; Dong, N.; Duquenne, P.-A.; Elsahar, H.; Gong, H.; Heffernan, K.; Hoffman, J.; et al","venue":"arXiv (Cornell University)","work_id":"9b58185d-8084-4802-9271-58d052a5af84","year":2023},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/2308.11596","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:7eedf14ada6823cbabd3d93ebe50a917f4c5cd3d746ddeb6b520b1c13fb902ee","observation_id":"5a506825-730e-487b-b13a-eb35d96fe3de","resolution":{"observed_at":"2026-05-17T03:18:57.169157Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Comet-22: Unbabel-ist 2022 submission for the metrics shared task","venue":null,"work_id":"1e3dfc82-4757-4ce6-9b79-09dc64c61cd8","year":2022},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:a5554580e03226a420c99c4bdd6bebd2a589bec4d1c9adf0f03fb65ee6766502","observation_id":"eb6292ec-d735-4e69-a737-51a20e6cc309","resolution":{"observed_at":"2026-05-17T03:23:58.421049Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A call for clarity in reporting BLEU scores","venue":null,"work_id":"0f6674d6-137d-4829-b494-fb0744b9663c","year":2018},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:7508375178de2f689a0b319c12ff86f451577892aa2c721f9097258dc3376b43","observation_id":"8090c14a-e813-413b-8ecc-60165eae2725","resolution":{"observed_at":"2026-05-17T03:23:58.415778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llamax: Scaling linguistic horizons of llm by enhancing translation capabilities beyond 100 lan- guages","venue":null,"work_id":"5ef40099-2aae-4ff9-abb7-c46673db496f","year":2024},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:4e06b317f702d6d86c569174f4c0a12a65479d6e21a061c9cb2d3b9ca4bc81e1","observation_id":"050cfcaa-e2b2-491b-b3d9-06f16413ead4","resolution":{"observed_at":"2026-05-17T03:23:58.448422Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20215","last_updated":"2025-03-26T04:17:55Z","snapshot_observed_at":"2026-08-06T08:46:20.194739Z","submitted_at":"2025-03-26T04:17:55Z","title":"Qwen2.5-Omni Technical Report","version":1},"cited_work":{"arxiv_id":"2503.20215","doi":"10.48550/arxiv.2503.20215","metadata_source":"pith","pith_arxiv_id":"2503.20215","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5-Omni Technical Report","venue":"cs.CL","work_id":"438f105c-fa9b-44aa-ad52-43acb8045cda","year":2025},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/2503.20215","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:6f7cd845473ef0a0a57a5207cae65f745ffe7479d373db4be88c5e4c66eece0c","observation_id":"a07dcb4a-4512-46d8-bcc0-f6ee6707d48e","resolution":{"observed_at":"2026-05-17T03:18:57.182742Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:18:44.887499+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:18:44.887499+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Efficient memory management for large language model serving with pagedattention","venue":null,"work_id":"9c9c39c9-30ac-4bb5-8684-ed9ebd9708f4","year":null},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:761a4f81107942c83c6abd106fccce0f290c2d5e9be5d81d53e9a203b073f3cb","observation_id":"041cebb7-80a8-44f8-a480-19ace663c866","resolution":{"observed_at":"2026-05-17T03:23:58.418482Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"225b630d-49e5-416d-ac72-2a8311ea7b68","year":2020},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:b21b6f07afcd7572cf7ec9b10365a455fbfb00f06aa75fda0eb53a668233674a","observation_id":"7fbb6436-ed49-4e9d-aad4-8ae41a139a08","resolution":{"observed_at":"2026-05-17T03:23:58.437877Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-30T01:31:44.078208Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages"},"reference_resolution":{"displayed":30,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":1,"verified_exact":14,"verified_fuzzy":14},"total_outbound_references":30},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 30 of 30 outbound references and 0 inbound Pith citation observations for arXiv:2512.01512."}