{"as_of":"2026-08-07T10:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:43bfe79df127b6378bd43e407ead4727153c1cb049d37d55a0fbe638768c545c","coverage":[{"denominator":106,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-16T21:49:21.785096Z","state":"measured"},{"denominator":106,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":106,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":6,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":6,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-29T13:34:02.718948Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-04T13:59:52.852722Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"cited_work":{"arxiv_id":"2512.16378","doi":null,"metadata_source":"pith","pith_arxiv_id":"2512.16378","snapshot_observed_at":"2026-07-04T13:59:52.852722Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","venue":"cs.CL","work_id":"d927cef5-e388-4aef-808c-1941fc461a0c","year":2025},"citing_paper":{"arxiv_id":"2601.02933","last_updated":"2026-04-20T07:29:30Z","snapshot_observed_at":"2026-08-02T22:15:22.544797Z","submitted_at":"2026-01-06T11:21:03Z","title":"Pearmut: Human Evaluation of Translation Made Trivial","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-16T17:55:47.263924Z"},"links":{"cited_paper":"/paper/2512.16378","citing_paper":"/paper/2601.02933"},"observation_digest":"sha256:0633b0276cc08a84a7eca88f3ea38a4fe8d0f36e9bceb85eacf0483202533c7f","observation_id":"ea353db9-35cd-49ee-b704-906a743bda08","resolution":{"observed_at":"2026-05-16T17:58:12.960698Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"cited_work":{"arxiv_id":"2512.16378","doi":null,"metadata_source":"pith","pith_arxiv_id":"2512.16378","snapshot_observed_at":"2026-07-04T13:59:52.852722Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","venue":"cs.CL","work_id":"d927cef5-e388-4aef-808c-1941fc461a0c","year":2025},"citing_paper":{"arxiv_id":"2605.28227","last_updated":"2026-05-27T09:41:22Z","snapshot_observed_at":"2026-08-06T20:40:41.928264Z","submitted_at":"2026-05-27T09:41:22Z","title":"Why We Need Speech to Evaluate Speech Translation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-29T13:34:02.718948Z"},"links":{"cited_paper":"/paper/2512.16378","citing_paper":"/paper/2605.28227"},"observation_digest":"sha256:d3dbf36ae4b777b845385eeb1359f565951affb40f0dc0be8f171be3668fa7ae","observation_id":"53248fd4-637a-472c-8e20-bd6d852c0914","resolution":{"observed_at":"2026-06-29T13:53:29.123598Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"cited_work":{"arxiv_id":"2512.16378","doi":null,"metadata_source":"pith","pith_arxiv_id":"2512.16378","snapshot_observed_at":"2026-07-04T13:59:52.852722Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","venue":"cs.CL","work_id":"d927cef5-e388-4aef-808c-1941fc461a0c","year":2025},"citing_paper":{"arxiv_id":"2606.06047","last_updated":"2026-06-04T11:42:37Z","snapshot_observed_at":"2026-07-06T23:45:56.266116Z","submitted_at":"2026-06-04T11:42:37Z","title":"Automatic Labelling of Speech Translation Errors","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-28T01:33:57.622162Z"},"links":{"cited_paper":"/paper/2512.16378","citing_paper":"/paper/2606.06047"},"observation_digest":"sha256:6b21ce2f3bbf734533a562f70605a78dfbe445dd0869cc59e1f003199ed99a1b","observation_id":"a959f7f4-b3f2-4e74-98b7-6fb993cedd9d","resolution":{"observed_at":"2026-07-02T13:06:59.393277Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"cited_work":{"arxiv_id":"2512.16378","doi":null,"metadata_source":"pith","pith_arxiv_id":"2512.16378","snapshot_observed_at":"2026-07-04T13:59:52.852722Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","venue":"cs.CL","work_id":"d927cef5-e388-4aef-808c-1941fc461a0c","year":2025},"citing_paper":{"arxiv_id":"2606.06177","last_updated":"2026-06-04T13:52:21Z","snapshot_observed_at":"2026-07-06T23:46:01.165261Z","submitted_at":"2026-06-04T13:52:21Z","title":"Ouvia: A User-centered Framework for Measuring Usability of Speech Translation in Real-World Communication Scenarios","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-06-28T01:11:56.037674Z"},"links":{"cited_paper":"/paper/2512.16378","citing_paper":"/paper/2606.06177"},"observation_digest":"sha256:f797f85e27b3a8783cdd583d657d16aa0569be8e509589044f35904627d3ede1","observation_id":"82746525-b4a9-4d28-a28a-e37c392a2c1b","resolution":{"observed_at":"2026-07-02T13:36:59.045963Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"cited_work":{"arxiv_id":"2512.16378","doi":null,"metadata_source":"pith","pith_arxiv_id":"2512.16378","snapshot_observed_at":"2026-07-04T13:59:52.852722Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","venue":"cs.CL","work_id":"d927cef5-e388-4aef-808c-1941fc461a0c","year":2025},"citing_paper":{"arxiv_id":"2606.17255","last_updated":"2026-06-15T19:57:05Z","snapshot_observed_at":"2026-07-06T23:52:52.701429Z","submitted_at":"2026-06-15T19:57:05Z","title":"MLLP-VRAIN UPV system for the IWSLT 2026 Simultaneous Speech Translation task","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-06-27T03:12:21.284880Z"},"links":{"cited_paper":"/paper/2512.16378","citing_paper":"/paper/2606.17255"},"observation_digest":"sha256:a0314812d9deca5424b764773ab736658232446c9cbfe251a432eb5c3d3ca729","observation_id":"f7b44eda-b9e5-4506-9d3e-c4db936988ad","resolution":{"observed_at":"2026-07-03T18:18:49.944800Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"cited_work":{"arxiv_id":"2512.16378","doi":null,"metadata_source":"pith","pith_arxiv_id":"2512.16378","snapshot_observed_at":"2026-07-04T13:59:52.852722Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","venue":"cs.CL","work_id":"d927cef5-e388-4aef-808c-1941fc461a0c","year":2025},"citing_paper":{"arxiv_id":"2606.26968","last_updated":"2026-06-25T12:40:39Z","snapshot_observed_at":"2026-07-31T01:20:28.481331Z","submitted_at":"2026-06-25T12:40:39Z","title":"RedVox: Safety and Fairness Gaps in Speech Models Across Languages","version":1},"reference_index":139,"source":"arxiv_source","source_observed_at":"2026-06-26T04:37:00.399470Z"},"links":{"cited_paper":"/paper/2512.16378","citing_paper":"/paper/2606.26968"},"observation_digest":"sha256:5be7af45586c06228dd5602bc83ecf365cc888d1e39819562f81a189d976e34d","observation_id":"8332c17a-75e7-4b89-b40b-19c0d2bcca4b","resolution":{"observed_at":"2026-07-04T13:59:52.853939Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2512.16378/citation-record","integrity":"/paper/2512.16378/integrity","json":"/paper/2512.16378/citation-record.json","paper":"/paper/2512.16378"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T17:53:58.945473Z","title":"URL: \" 'urlintro :=","venue":null,"work_id":"f09f9a4d-de00-45d0-ab05-eb62b1712f20","year":null},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:2135ebb7441e679b8abd5ae463c20bdf69b1ec180a87979a24e3175f17ef911a","observation_id":"10f01fc1-a930-46ec-a14d-fff52dde6fa1","resolution":{"observed_at":"2026-05-16T21:51:18.233843Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"write newline","venue":null,"work_id":"4a4861c3-169e-4340-a548-3c2bac55b26c","year":null},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:61e70de3381f717bb7eb2094052b4cb3af71f6a332c64997a863c113bbb024ce","observation_id":"8e47acee-1d76-48a6-863f-a02efd8ca0e1","resolution":{"observed_at":"2026-05-16T21:51:18.235871Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.iwslt-1.44","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"u ker, Katsuhito Sudoh, Brian Thompson, Marco Turchi, Alex Waibel, Patrick Wilken, Rodolfo Zevallos, Vil \\'e m Zouhar, and Maike Z \\","venue":null,"work_id":"115d1eed-a49d-4f3a-ada3-5f2eaecae5d4","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:5da471fef1dde6fc257ec75b493bc9620dba41d0c4cc7b0417c0cf96312d76d3","observation_id":"f15c26e7-28e5-4547-9825-edf73625c82d","resolution":{"observed_at":"2026-05-16T21:51:17.493870Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01743","last_updated":"2025-03-07T09:05:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-03T17:05:52Z","title":"Phi-4-Mini Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs","version":2},"cited_work":{"arxiv_id":"2503.01743","doi":"10.18653/v1/2023.wmt-1.23","metadata_source":"pith","pith_arxiv_id":"2503.01743","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Phi-4-Mini Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs","venue":"cs.CL","work_id":"83956045-536a-41ff-af02-b80e2a614eab","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2503.01743","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:e7a0b927157172f514956e10dbc8ad3218ac492a581958c3e5e8a7143abf3f76","observation_id":"1aa76f50-759a-4e00-9a4a-da4a95f0bf81","resolution":{"observed_at":"2026-05-16T21:51:17.818489Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:a747fc94922a3f47cf03a50c12480df35fe2c64e35e3be9d9a75c18695fedaea","observation_id":"f6caaff2-1bac-4d50-83d4-aa33e7d17438","resolution":{"observed_at":"2026-05-16T21:51:17.822086Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.iwslt-1.1","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"FINDINGS OF THE IWSLT 2024 EVALUATION CAMPAIGN","venue":null,"work_id":"8b7de965-c19e-4eb5-8f20-6c0222e94b02","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:872ff9b7e2d109437fcdc11d778d8e3574365de6c7b17fbee9c37e40abf5769d","observation_id":"a31477b0-2bac-4a6e-967c-1a1dc54a2766","resolution":{"observed_at":"2026-05-16T21:51:17.496302Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.emnlp-main.298","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GQA : Training generalized multi-query transformer models from multi-head checkpoints","venue":null,"work_id":"eadb5933-87f2-4102-94c8-52fe2ccbc1e2","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:46b9f4942e8b7bb96c429aab09de68f430e82ca6a35ccc32e48ee89b05d01012","observation_id":"563bb098-9256-4254-97d5-6b26931b211b","resolution":{"observed_at":"2026-05-16T21:51:17.491671Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-13T23:49:40.00604+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T23:49:40.00604+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b9e52d42-48ae-443c-adfd-3f67fb651c39","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:acb6b85663701320a28b76e6b207a778755c91d988c4bbe08a7c873ee6105f6d","observation_id":"acdfb7e3-fe90-456a-af2b-9b9891271080","resolution":{"observed_at":"2026-05-16T21:51:18.251322Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2503.10620","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T13:59:52.836168Z","title":"From tower to spire: Adding the speech modality to a text-only llm.arXiv preprint arXiv:2503.10620, 2025","venue":null,"work_id":"f269ec46-c502-4ea5-a125-af5794b013c4","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:24a69e3f54b0c1e60c8f574e84a69615ef9e1c5509a312dbe2cda3bf651816b3","observation_id":"300252cb-c103-42af-95b3-da7dadb58108","resolution":{"observed_at":"2026-05-16T21:51:17.801798Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2020.iwslt-1.1","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"08e6fec3-8012-465e-be0d-4e370a6a7268","year":2020},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:27db96598bb2ec6845e79bf8c8abff6ee7b5e7ed3ab9ea55cb72d21b67ed944d","observation_id":"fbe10fbc-710e-439f-9704-9b487aba7a56","resolution":{"observed_at":"2026-05-16T21:51:17.510429Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21437/interspeech.2023-2279","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"c9415401-2747-4f92-8bec-2b0bcb8152df","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:18be77017cd04abfb298b19d793411fab1c2aa22f05785bc39e70701708be197","observation_id":"f1626ba7-622f-4be7-8c1e-a4c72ec7b879","resolution":{"observed_at":"2026-05-16T21:51:17.539824Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"007fd936-983a-4ef3-a0a8-dcd82e1dd87c","year":2020},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:54e4a0fefb205b50c4f53d2cf819a030e8330344d441116464b322cfc07b4c6e","observation_id":"9e8530ea-6ab7-45f3-ac63-59ce3a2b1375","resolution":{"observed_at":"2026-05-16T21:51:18.242731Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.emnlp-main.1188","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"075e6cb5-373c-49e4-88fd-b8cac2d75eaf","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:0bfcc7985a9e0ea1d039d83ae82eba2398f3b138ec71811143ad3f52dc5ba7db","observation_id":"1080224f-7e85-4bb1-a36b-8df05505366b","resolution":{"observed_at":"2026-05-16T21:51:17.580251Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"97a2df58-a23b-4d0e-8fe5-83ff79c682f8","year":2020},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:3f6e836a5317d86c574971c036577aff67a82557ee2a95ea059989993830ee67","observation_id":"b0c998d8-5e3a-4b71-b4c1-0d3f751eb924","resolution":{"observed_at":"2026-05-16T21:51:18.223541Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":"2309.16609","doi":"10.48550/arxiv.2309.16609","metadata_source":"pith","pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen Technical Report","venue":"cs.CL","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:067f00a37f78dd42b29cfc824e3a1f6226392ca09664f7481a6bb6a3525b35f9","observation_id":"149b62cc-939c-4ef1-81e2-f45b7e4b3c6c","resolution":{"observed_at":"2026-05-16T21:51:17.782258Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2202.01374","last_updated":"2022-02-03T02:26:40Z","snapshot_observed_at":"2026-07-06T12:34:00.262764Z","submitted_at":"2022-02-03T02:26:40Z","title":"mSLAM: Massively multilingual joint pre-training for speech and text","version":1},"cited_work":{"arxiv_id":"2202.01374","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2202.01374","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bapna, C","venue":null,"work_id":"c143c52c-596d-4470-b0f4-abac681bd925","year":2022},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2202.01374","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:0403713cc219907373a00a692eaefc9e0ceffa6263cf1499985fb2d47312c9ff","observation_id":"d12b6ab0-06fa-4b7e-8094-9df5092d3073","resolution":{"observed_at":"2026-05-16T21:51:17.754099Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11596","last_updated":"2023-10-25T03:52:07Z","snapshot_observed_at":"2026-07-06T16:09:09.018523Z","submitted_at":"2023-08-22T17:44:18Z","title":"SeamlessM4T: Massively Multilingual & Multimodal Machine Translation","version":3},"cited_work":{"arxiv_id":"2308.11596","doi":"10.48550/arxiv.2308.11596","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.11596","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"C.; Dale, D.; Dong, N.; Duquenne, P.-A.; Elsahar, H.; Gong, H.; Heffernan, K.; Hoffman, J.; et al","venue":"arXiv (Cornell University)","work_id":"9b58185d-8084-4802-9271-58d052a5af84","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2308.11596","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:972005ccc65e21a421741551e1d67459b8579971a532909df0b5b052c98cbaed","observation_id":"57052b23-d293-4702-997e-55dff8335d0d","resolution":{"observed_at":"2026-05-16T21:51:17.767713Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a9e89f1c-2ac7-4f01-8fe3-3fd0ca96da91","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:879616b4fd46c0f1fd4a7441317f4fdbe70e99b03ba5a915d1c2dc3559d92f87","observation_id":"9912d184-ffe9-43a9-b5e2-68655fafffae","resolution":{"observed_at":"2026-05-16T21:51:18.244591Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2021.acl-long.224","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"3eec9092-eb3a-4285-8b2c-23bb285be040","year":2021},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:2d1d844e6f88b6236ca10f17215297af80c8c18a057c9a95c312ec6b4d9244cf","observation_id":"9ebd1d40-49be-4f96-a33e-f8735979ea93","resolution":{"observed_at":"2026-05-16T21:51:17.585065Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1612.01744","last_updated":"2016-12-06T10:48:56Z","snapshot_observed_at":"2026-07-06T05:21:31.168402Z","submitted_at":"2016-12-06T10:48:56Z","title":"Listen and Translate: A Proof of Concept for End-to-End Speech-to-Text Translation","version":1},"cited_work":{"arxiv_id":"1612.01744","doi":null,"metadata_source":"pith","pith_arxiv_id":"1612.01744","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Listen and Translate: A Proof of Concept for End-to-End Speech-to-Text Translation","venue":"cs.CL","work_id":"14c2629f-f543-4909-b9ef-da1ed7fa8530","year":2016},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/1612.01744","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:193baef3f393bab8971b7753dbd0874e0d87480b8c8a7f8e264bd3c8c7b26217","observation_id":"bab6c848-266e-4810-9c0d-66fbb21b00e0","resolution":{"observed_at":"2026-05-16T21:51:17.794918Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10759","last_updated":"2024-07-15T14:38:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T14:38:09Z","title":"Qwen2-Audio Technical Report","version":1},"cited_work":{"arxiv_id":"2407.10759","doi":"10.48550/arxiv.2407.10759","metadata_source":"pith","pith_arxiv_id":"2407.10759","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2-Audio Technical Report","venue":"eess.AS","work_id":"c249e63c-cf40-408f-a4ff-fdf68e8cbeb8","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2407.10759","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:77f1a6126ae09938dc49d82003e5db0326365e2f3c2d0e1b05d67ef64b092373","observation_id":"16cd986a-493b-47ef-bad2-9f61a89a480d","resolution":{"observed_at":"2026-05-16T21:51:17.762720Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-07-06T13:13:37.143547Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:2bc6cc309580482c2e4d2c3cf1b9d7d36790dfeb0410383c17e056aa45c9a044","observation_id":"f47c68f1-7d4f-4085-afb9-b4dd2c8aedce","resolution":{"observed_at":"2026-05-16T21:51:17.797958Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Costa-juss \\`a , Christine Basta, and Gerard I","venue":null,"work_id":"1c630c80-5908-4236-b5a9-bbc201e59e74","year":2022},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:2898a88650f3e6ae7bdac3ffb44041722b67a801cae883a6ca99dca957eb45dd","observation_id":"a8feccc2-3190-41ef-9766-4cd6337da12b","resolution":{"observed_at":"2026-05-16T21:51:18.249606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.04672","last_updated":"2022-08-25T17:10:53Z","snapshot_observed_at":"2026-07-06T13:29:47.927628Z","submitted_at":"2022-07-11T07:33:36Z","title":"No Language Left Behind: Scaling Human-Centered Machine Translation","version":3},"cited_work":{"arxiv_id":"2207.04672","doi":"10.18653/v1/w19-5207","metadata_source":"pith","pith_arxiv_id":"2207.04672","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"No Language Left Behind: Scaling Human-Centered Machine Translation","venue":"cs.CL","work_id":"68c8336c-d20e-40fa-ba9c-89459da6fc1a","year":2022},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2207.04672","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:9d0eb14fc728d089a22a436149497a77ad333c8f1b915554fea6ce39316d67c1","observation_id":"04364b94-1598-44b1-ba04-2ec9359c6b2c","resolution":{"observed_at":"2026-05-16T21:51:17.788392Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04261","last_updated":"2024-12-05T15:41:06Z","snapshot_observed_at":"2026-07-06T20:02:17.096487Z","submitted_at":"2024-12-05T15:41:06Z","title":"Aya Expanse: Combining Research Breakthroughs for a New Multilingual Frontier","version":1},"cited_work":{"arxiv_id":"2412.04261","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.04261","snapshot_observed_at":"2026-07-04T16:29:57.395027Z","title":"DeepSeek-AI, D","venue":null,"work_id":"2fe6961d-2ae4-4175-8400-b94838b20783","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2412.04261","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:5c3c8f1b43626eb185679528978ee3b21f92820cb4bcd48b2601b458b3daa275","observation_id":"00129a49-c0bb-459c-9987-021ece5230be","resolution":{"observed_at":"2026-05-16T21:51:17.785347Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.findings-acl.634","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Findings of the Association for Computational Linguistics: ACL 2025 , month=","venue":null,"work_id":"faa67860-72d8-4371-af3b-7c96278e097b","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:3b73b0fac6c6026124ee9fdf6adfbf9ad25e9a662ef221dccaf887008d026df2","observation_id":"e346d74d-5ff2-40d2-9076-979e830f42e7","resolution":{"observed_at":"2026-05-16T21:51:17.586352Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.wmt-1.51","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Results of WMT 23 Metrics Shared Task: Metrics Might Be Guilty but References Are Not Innocent","venue":null,"work_id":"3e95c7c7-7494-457b-8a30-1774d39f183c","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:9296d2d499410418091ef25d4f74a5d6c45a3f0c5e6dd26de14bb8d968555219","observation_id":"79308b10-b4eb-496e-affe-fac5d9e8e3e0","resolution":{"observed_at":"2026-05-16T21:51:17.537696Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.acl-long.789","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"7a062b91-4866-454f-82e4-7a988d822693","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:a816f933e0effd54259ce39e201fe6d3b4cbf7d7270f8da66b8b8c2f81a453d8","observation_id":"b9af18a6-d0f3-433b-a1ff-bdda173ecdaf","resolution":{"observed_at":"2026-05-16T21:51:17.582962Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2021.emnlp-main.128","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing","work_id":"ed6be0fc-5fa8-4afd-8684-0dafe770b71f","year":2021},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:d7516a218d89d31f7b98d40f05ad50f541caf66c10201c04e217cb21f0924463","observation_id":"c9792b03-c2e8-47b9-ab8e-2a28dcc548c2","resolution":{"observed_at":"2026-05-16T21:51:17.603327Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"76ed0adc-4dcd-4d17-93ce-959be6855372","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:530a61688d04082b2d240d83d768b509ee8ab89d00b79348a17cb389332b272b","observation_id":"2510294d-435c-423a-8d73-28a4ce3445f1","resolution":{"observed_at":"2026-05-16T21:51:18.247983Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19786","last_updated":"2025-03-25T15:52:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-25T15:52:34Z","title":"Gemma 3 Technical Report","version":1},"cited_work":{"arxiv_id":"2503.19786","doi":"10.1007/978-3-540-48085-3_36","metadata_source":"pith","pith_arxiv_id":"2503.19786","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemma 3 Technical Report","venue":"cs.CL","work_id":"f93e08bf-9e96-409b-8ac6-b8385fd17fd7","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2503.19786","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:4cb7894724378449a63b1fa70412e2c73af55abe531e05a5222f4712fa7eb84f","observation_id":"62c86786-e8e3-489a-aee4-6ba80cba1597","resolution":{"observed_at":"2026-05-16T21:51:17.808565Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":"2407.21783","doi":"10.1016/s0749-0720(15","metadata_source":"pith","pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"The Llama 3 Herd of Models","venue":"cs.AI","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:931bf84388bd6eead525a33d5dff5412c0029a8b134eb6e7def53775218581ad","observation_id":"b970ba17-76d4-4002-ba5d-a32877ba1961","resolution":{"observed_at":"2026-05-16T21:51:17.779391Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21437/interspeech.2020-3015","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi:10.21437/Interspeech.2020-3015 , issn =","venue":null,"work_id":"252d779f-5a82-47a1-80e9-d0e5228e48d8","year":2020},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:fdaebdc909dd0f8bdc5e5c58d32d9066b27a991d97b854db5b4da661c7003be6","observation_id":"f07e9987-2ecd-46c9-b12c-de6d1441dd93","resolution":{"observed_at":"2026-05-16T21:51:17.606530Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-11T13:49:10.351598+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T13:49:10.351598+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04792","last_updated":"2024-02-29T20:59:17Z","snapshot_observed_at":"2026-08-07T05:40:44.041304Z","submitted_at":"2024-02-07T12:31:13Z","title":"Direct Language Model Alignment from Online AI Feedback","version":2},"cited_work":{"arxiv_id":"2402.04792","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04792","snapshot_observed_at":"2026-07-04T11:09:46.394396Z","title":"Direct language model alignment from online ai feedback","venue":null,"work_id":"55e38714-7a1a-46a1-81c1-04bccd2847f4","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2402.04792","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:5d5b53bd08e23255da52bcbee55a73edc443588fe0d34e9782348e7231aca5f1","observation_id":"a917cb65-00e6-4af7-8631-da3807faf630","resolution":{"observed_at":"2026-05-16T21:51:17.792019Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.emnlp-main.1218","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"35eb47f2-2090-4cc2-b9f9-e302761bbdcd","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:ebf67051d65892644b7f9600cff8bdc1a295219c6536a683c4a6978a7c83b245","observation_id":"dc878bb1-102b-4602-9e6b-46864ab706b5","resolution":{"observed_at":"2026-05-16T21:51:17.537940Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.sigmorphon-main.3","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"4ff18ff2-7dee-4653-9843-9a02178029de","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:0af2830a0d595f0ba12228a917a3679870ee91f5834c91d1ffccc6080bb90672","observation_id":"d7936d8e-20de-464c-952a-80e972480ccd","resolution":{"observed_at":"2026-05-16T21:51:17.591425Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2021.312229","doi":"10.1109/taslp.2021.3122291","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"HuBERT: Self-supervised speech representation learning by masked prediction of hidden units","venue":"IEEE/ACM Transactions on Audio Speech and Language Processing","work_id":"01bc0841-3ef8-4720-9eb0-fe0a5a831e62","year":2021},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:2238cf11cb8756c52a15bb276c9601df8d76e7f3274026ef4331a78d82e06010","observation_id":"347ac498-5369-4340-be51-37aadad64412","resolution":{"observed_at":"2026-05-16T21:51:17.622857Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-11T04:49:33.939513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T04:49:33.939513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"344963f1-ac4e-440c-b4a8-1e8fc22bb401","year":2022},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:9162f1fe7351fce7611ec563e1fe47ad953c969e5a1ac1d46976a409a01b3c9c","observation_id":"c3c6c6bb-5d4e-42ea-81a8-61e7370504dd","resolution":{"observed_at":"2026-05-16T21:51:18.238784Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"0776.2020","doi":"10.1109/icassp40776.2020","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Zheng, C","venue":null,"work_id":"714cc1a7-8c25-492d-be79-11b0b30f6eb9","year":2020},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:f896589bc37c262aaec762b551e58533db4f4b0a0bda87c2cc41e9ea0be31017","observation_id":"6dbcd16c-8d8c-4097-bef8-be1e2a116c67","resolution":{"observed_at":"2026-05-16T21:51:17.602153Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Iranzo-Sánchez , J","venue":null,"work_id":"83fe36ce-ff47-4ac6-a83a-23aac14a574c","year":2020},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:6a454095e98d4cfcefdb74ae3a19a021e57171b7ebcbeb31fff5b045fc1d5c03","observation_id":"7bcd0085-06d7-4977-bdc1-63f1b1f47d27","resolution":{"observed_at":"2026-05-16T21:51:18.240796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21437/interspeech.2022-10938","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":"Interspeech 2022","work_id":"5371f4ad-c0ea-49e6-81f2-ea5bb1c815e7","year":2022},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:05527ae533f8b03b0537a595bfeec8bebf6eadd6977e77190ecfece965e55787","observation_id":"29b69461-74f0-45c4-a2c8-761b791b4060","resolution":{"observed_at":"2026-05-16T21:51:17.578021Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.wmt-1.35","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"M etric X -24: The G oogle Submission to the WMT 2024 Metrics Shared Task","venue":null,"work_id":"c6c5c00a-4d4d-4920-b43c-d30fe2ad5354","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:5cd659c5657cf6033ae765c8973cad872eed69c93c77953dfdeb3a4ca51359e5","observation_id":"a60025f1-a714-4cdf-879f-721d0ea64cea","resolution":{"observed_at":"2026-05-16T21:51:17.611943Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b27bd387-8e67-4a29-ad14-3c333cffbb1b","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:10dd9b3ce39593b1fa05c2aab96c6c38d7c927c81c4bccb27681e85b10dfe3be","observation_id":"e2172a71-5641-45a2-979a-7b83b91bfa3f","resolution":{"observed_at":"2026-05-16T21:51:18.236818Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.wmt-1.1","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Findings of the","venue":null,"work_id":"cb1d17b3-9074-46cb-9920-964d6351ad1c","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:1654e61f9ceb580e3403f08455b472fd0a454410797b162fdcd15a42710b92ec","observation_id":"a2c59bb6-a73b-4b0f-b5c0-a7411475174a","resolution":{"observed_at":"2026-05-16T21:51:17.513039Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-05-22T13:52:41.61172+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-22T13:52:41.61172+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.wmt-1.131","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Error Span Annotation: A Balanced Approach for Human Evaluation of Machine Translation","venue":null,"work_id":"f5886061-fced-4a64-a821-306f5332364e","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:d2a1af4e0d45f679a86a977bc4f07e18f20e36ddfca0624c006ee0eeb24ba513","observation_id":"b5d0e3d2-c8b0-44ba-8b30-56f0caddfee7","resolution":{"observed_at":"2026-05-16T21:51:17.615540Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-05-21T06:52:40.915481+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-21T06:52:40.915481+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.acl-long.110","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"0e198665-995d-46dd-acaa-8a974f0b9996","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:fb8d0e06c42ee5716631238ea41ed9767a4bb9646f3e6383b8859aa33d1726fb","observation_id":"d2b0650a-ea37-4625-bd94-7df017a4dd14","resolution":{"observed_at":"2026-05-16T21:51:17.561031Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.iwslt-1.22","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"5436c232-f817-41d5-adb7-d2262f214132","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:9b90360fc7c0d479310251c9341aa96ddd3a242dab51ca265e2abaeeee6db69a","observation_id":"66cb3f8d-7ada-46f3-8480-1299384539ed","resolution":{"observed_at":"2026-05-16T21:51:17.598298Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2269.361559","doi":"10.1145/3582269.3615596","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gender bias and stereotypes in Large Language Models , url=","venue":null,"work_id":"fde64aa1-e553-4a96-ac4a-612957d594a3","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:417bc6753b74c165242bb017fc77ffd8c982467dec2c908d785a573eaa66ab84","observation_id":"16b01789-4d96-4081-818b-22678cdbf7c5","resolution":{"observed_at":"2026-05-16T21:51:17.567779Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.09577","last_updated":"2019-09-14T03:51:46Z","snapshot_observed_at":"2026-08-03T18:17:55.398298Z","submitted_at":"2019-09-14T03:51:46Z","title":"NeMo: a toolkit for building AI applications using Neural Modules","version":1},"cited_work":{"arxiv_id":"1909.09577","doi":null,"metadata_source":"pith","pith_arxiv_id":"1909.09577","snapshot_observed_at":"2026-07-08T17:15:09.332377Z","title":"arXiv preprint arXiv:1909.09577 , year=","venue":"cs.LG","work_id":"8025af67-043c-453f-b000-17014649401b","year":2019},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/1909.09577","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:a922e7589033bc65f32d43e27db992259737a55fd06392678b519351f27499d9","observation_id":"ce91664c-ed95-4171-b71c-eec430d6f3f8","resolution":{"observed_at":"2026-05-16T21:51:17.764549Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1808.06226","last_updated":"2018-08-19T16:49:06Z","snapshot_observed_at":"2026-07-06T06:56:20.935388Z","submitted_at":"2018-08-19T16:49:06Z","title":"SentencePiece: A simple and language independent subword tokenizer and detokenizer for Neural Text Processing","version":1},"cited_work":{"arxiv_id":"1808.06226","doi":"10.18653/v1/d18-2012","metadata_source":"doi_reference","pith_arxiv_id":"1808.06226","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Sentencepiece: A simple and language independent subword tokenizer and detokenizer for neural text processing","venue":"cs.CL","work_id":"81a6320b-c2e1-4d74-a03e-9e1ff6bbed8d","year":2018},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/1808.06226","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:4494de326a53f645f86eaa8c84cba6c584e1b0e233450615807981ce0030d3d6","observation_id":"2b90a4be-7444-495c-85f3-664e7c25e3b2","resolution":{"observed_at":"2026-05-16T21:51:17.775781Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-11T12:19:12.876508+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T12:19:12.876508+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12792","last_updated":"2023-09-22T01:44:46Z","snapshot_observed_at":"2026-08-04T07:24:03.964437Z","submitted_at":"2023-08-24T13:47:16Z","title":"Sparks of Large Audio Models: A Survey and Outlook","version":3},"cited_work":{"arxiv_id":"2308.12792","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.12792","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Schuller","venue":null,"work_id":"a75066e0-be80-49f3-adf5-c74af9ac4b8d","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2308.12792","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:fe8a3f2b3bc80d63d5ec0721fcb35cf83a01b79ec57e4d4b407e6d6a9a5a1d47","observation_id":"1b5d64a3-e72e-4c99-85af-ec6faa9844d4","resolution":{"observed_at":"2026-05-16T21:51:17.776305Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.wmt-1.24","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"In: Haddow, B., Kocmi, T., Koehn, P., Monz, C","venue":null,"work_id":"44bf3b3d-3efb-4c64-9fa9-7adc0533224d","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:b85f306f1a2518e4de25191c7419dd80a252592b39ab0c072efa303038541f71","observation_id":"080f949c-2d24-4030-8f5f-cd199aec32fc","resolution":{"observed_at":"2026-05-16T21:51:17.547148Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-05-22T13:52:45.103129+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-22T13:52:45.103129+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T16:21:16.184360Z","title":null,"venue":null,"work_id":"df1729dc-5393-46dc-9c0d-1606cfa191fa","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:fa4d089b2570f1659e2e3b7a28d59eb2f606879ccbe1b4c2bd1fedafbbf7cc84","observation_id":"93cfe5a3-5bb5-48b9-8d70-6d87b5ab1520","resolution":{"observed_at":"2026-05-16T21:51:18.235098Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.13264","last_updated":"2025-07-17T16:17:37Z","snapshot_observed_at":"2026-08-06T16:24:15.622905Z","submitted_at":"2025-07-17T16:17:37Z","title":"Voxtral","version":1},"cited_work":{"arxiv_id":"2507.13264","doi":"10.48550/arxiv.2507.13264","metadata_source":"pith","pith_arxiv_id":"2507.13264","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V oxtral","venue":"cs.SD","work_id":"d8af79fd-31af-429f-b7c7-b98cea5e46ed","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2507.13264","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:7b04bd306f3cd734d56cf5873dd40be770c40cc95a9457d962865249e087ab88","observation_id":"aedec695-5e30-49da-8b5f-1e495ca7d192","resolution":{"observed_at":"2026-05-16T21:51:17.744592Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21437/interspeech.2023-2528","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"bbbe9203-fc20-4687-aa22-5663f1ab38d2","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:8915c62b2a35aeac048bbe3a18913efac8921287f4b1a296a662945e19e771f0","observation_id":"f8889288-5ee5-4621-bed5-66b0310d5738","resolution":{"observed_at":"2026-05-16T21:51:17.572989Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.20007","last_updated":"2025-01-27T08:46:09Z","snapshot_observed_at":"2026-08-05T02:18:00.592980Z","submitted_at":"2024-09-30T07:01:21Z","title":"DeSTA2: Developing Instruction-Following Speech Language Model Without Speech Instruction-Tuning Data","version":2},"cited_work":{"arxiv_id":"2409.20007","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.20007","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"DeSTA: Enhancing speech language models through descriptive speech-text alignment","venue":null,"work_id":"1eb2e8bb-0c45-4a4c-ae57-c3ab39f34400","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2409.20007","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:484af2399db057ae16905979d6b89eafc4867eea81622460d958f484fc71c0d8","observation_id":"784b4ec4-77fc-4888-923c-2e56cdc2d81e","resolution":{"observed_at":"2026-05-16T21:51:17.771428Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.iwslt-1.12","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"811b3d62-aa4e-45c5-90d2-b661df897fde","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:84721886c243d6f219659d8e18e9d5ed0709c2b370cf15c407e9d0185115dd6f","observation_id":"99efa011-e5a9-42da-8dc5-ea81da485327","resolution":{"observed_at":"2026-05-16T21:51:17.575456Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c86f9da3-47df-40fb-a2e8-e463b29770bd","year":2005},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:ee31dd95f8c4d9618ba57049a0b9f92839a7882389897a2f800097f9066109b8","observation_id":"7aad855f-0dfc-40e1-b091-69054360ce87","resolution":{"observed_at":"2026-05-16T21:51:18.233440Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.00377","last_updated":"2025-02-01T09:29:21Z","snapshot_observed_at":"2026-08-01T19:26:06.149058Z","submitted_at":"2025-02-01T09:29:21Z","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","version":1},"cited_work":{"arxiv_id":"2502.00377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.00377","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a2294b94-d6ed-481c-b90f-1ad6868c8654","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2502.00377","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:3b0cdf8a3493cb62c78b34ce44564a045226226205848c4b8a00559a567eadce","observation_id":"3f40e10f-5ebd-426b-94cd-250e0379a231","resolution":{"observed_at":"2026-05-16T21:51:17.766208Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"1999.758176","doi":"10.1109/icassp.1999.758176","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"c3063aae-a352-4ebd-8dff-017dccce3223","year":1999},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:b92d114d536ec80bf9f498b9d460787cfee89206e318c65dac152e937d9e2fa5","observation_id":"6aed4b5e-a804-4896-885f-0ad35881ed21","resolution":{"observed_at":"2026-05-16T21:51:17.606698Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"0776.2020","doi":"10.1109/icassp40776.2020","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Zheng, C","venue":null,"work_id":"714cc1a7-8c25-492d-be79-11b0b30f6eb9","year":2020},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:915175c92fd46607c82429c5c10d399baa1ad15564b4f2250631e285f3194cf9","observation_id":"447bca27-a0ca-4e09-99ab-8302b15078f6","resolution":{"observed_at":"2026-05-16T21:51:17.619454Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21437/interspeech.2023-1905","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2023 , booktitle =","venue":null,"work_id":"52ba3005-106a-443d-ae07-564dcdcb5bfa","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:d7807184fe098bc4c0340fbdd3c6567ca5d9f9c08a54d8c98c4665809e7469be","observation_id":"aea33282-9ca4-4f09-a76b-701ab5c0addf","resolution":{"observed_at":"2026-05-16T21:51:17.625250Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1162/tacl_a_00728","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T02:25:14.856196Z","title":null,"venue":null,"work_id":"2f5e2c02-c52a-4aa3-bbd7-b0c3b254c71e","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:1be4936f2e6baa2c8cb03985ef7728f989649feeb440f865cd7af28d5d80392e","observation_id":"1afab664-b610-4e8f-b144-7c62838a30d1","resolution":{"observed_at":"2026-05-16T21:51:17.593478Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2015.717896","doi":"10.1109/icassp.2015.7178964","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"& Khudanpur, S","venue":null,"work_id":"fff28059-39d2-4749-b4dd-948a4512606e","year":2015},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:3db5be44677e565f848280ecc7452550f66660671e42396f4f471ac9fb135c22","observation_id":"c1c80036-bee2-4a48-973b-5c08001b04b9","resolution":{"observed_at":"2026-05-16T21:51:17.570609Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-11T13:49:10.124279+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T13:49:10.124279+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.03172","last_updated":"2023-11-20T23:09:34Z","snapshot_observed_at":"2026-07-06T15:51:12.179086Z","submitted_at":"2023-07-06T17:54:11Z","title":"Lost in the Middle: How Language Models Use Long Contexts","version":3},"cited_work":{"arxiv_id":"2307.03172","doi":"10.1162/tacl","metadata_source":"pith","pith_arxiv_id":"2307.03172","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Lost in the Middle: How Language Models Use Long Contexts","venue":"cs.CL","work_id":"37c05e13-4a24-44f8-a1c4-da1bbe7223aa","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2307.03172","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:ef279e7660f230b30c915fd92ac9dd6933f5a8404ef3349667fc90c3265d7601","observation_id":"4ec30b38-635f-4f9f-894e-8418ab8b4471","resolution":{"observed_at":"2026-05-16T21:51:17.597732Z","resolver_source":"doi_truncated","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.19634","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mcif: Multimodal crosslingual instruction-following benchmark from scientific talks","venue":null,"work_id":"6a67230f-1bb6-4191-9d19-576c031cd15e","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:eab0c208b484baf482347de52b69c72f8cecb803d5fbb383f17f374d923d700c","observation_id":"d44a3ee0-9415-4c8a-9e00-e6af23bad6be","resolution":{"observed_at":"2026-05-16T21:51:17.788247Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21437/interspeech.2025-1062","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OWSM v4: Improving Open Whisper-Style Speech Models via Data Scaling and Cleaning","venue":null,"work_id":"f4aa33a1-5d2d-4ace-bc57-5a7db7564026","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:a61de44065b7a292cc01f295621d5da09d282cf28d281d6593ad1e69dd482022","observation_id":"f4f3123c-d187-45eb-89fa-e00334dfca0a","resolution":{"observed_at":"2026-05-16T21:51:17.588671Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"McCarthy, and Deepak Gopinath","venue":null,"work_id":"53f4d3ce-87a7-41e1-8458-a6792287ffad","year":2019},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:214c34171b6c06f26d4b5db3ac3b3aa16edbfbc2ab839562418a9f04813c33df","observation_id":"dc871a4d-2822-49a6-8677-3821192f39d1","resolution":{"observed_at":"2026-05-16T21:51:18.231366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T01:13:13.729736Z","title":null,"venue":null,"work_id":"cf0695b2-8b90-4b1e-9e6d-29f9da4f0bfa","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:67191ad0ad4544e8398de89187d0f8f654bc430f694e85018e2ede5fe52aa389","observation_id":"ce7069e8-f309-4553-9f9a-b5af26570991","resolution":{"observed_at":"2026-05-16T21:51:18.222483Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T05:23:20.908461Z","title":null,"venue":null,"work_id":"0cf5d12c-a56e-40d6-8cce-73c78b9e53ec","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:e62b70beebe4fff62fe88b83c5f93d0a3fa96c5a32d856435e10334d51c3f1ce","observation_id":"0162e3a5-9e2d-4f2e-8044-8cf446c1a309","resolution":{"observed_at":"2026-05-16T21:51:18.224343Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21437/interspeech.2025-540","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"57c48695-7023-4357-b326-a495971ad6f4","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:76d4913387e98e9fa54172e88518b00cf83690a978bc76a99da23031335df71a","observation_id":"100c6892-d660-4100-b166-4bbefe47fc16","resolution":{"observed_at":"2026-05-16T21:51:17.621584Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.17080","last_updated":"2025-06-20T15:30:06Z","snapshot_observed_at":"2026-08-06T23:28:40.905407Z","submitted_at":"2025-06-20T15:30:06Z","title":"Tower+: Bridging Generality and Translation Specialization in Multilingual LLMs","version":1},"cited_work":{"arxiv_id":"2506.17080","doi":"10.48550/arxiv.2506.17080","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.17080","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"https://arxiv.org/abs/2506.17080","venue":"ArXiv.org","work_id":"59371777-0482-4d20-858b-756f93c6c887","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2506.17080","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:7fd88ff8b8c622acc8e18d41d6a4166283e1e293fcf49c2991825121e058a039","observation_id":"fb3eaccb-36ba-45d4-b9ab-e375cd14605e","resolution":{"observed_at":"2026-05-16T21:51:17.791820Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1c498329-ec41-44f5-84d8-6f7ff6a69ade","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:e2f425cc464e025e95017886e315b85d84292ed25e0b06e50e2214508923861e","observation_id":"e4fb823f-b766-40b1-92c9-220c0df844ab","resolution":{"observed_at":"2026-05-16T21:51:18.228292Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2022.iwslt-1.30","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"874287f8-b570-4028-b21b-3a9bb20167f2","year":2022},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:ebb635da28808680d2c60a0c5686286017c0d039c63a0330efaf4b08d2073028","observation_id":"79234645-a579-472a-9a17-7c7bb7d82e72","resolution":{"observed_at":"2026-05-16T21:51:17.521336Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.12925","last_updated":"2023-06-22T14:37:54Z","snapshot_observed_at":"2026-08-07T01:18:02.068918Z","submitted_at":"2023-06-22T14:37:54Z","title":"AudioPaLM: A Large Language Model That Can Speak and Listen","version":1},"cited_work":{"arxiv_id":"2306.12925","doi":"10.21437/interspeech.2019-1873","metadata_source":"pith","pith_arxiv_id":"2306.12925","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AudioPaLM: A Large Language Model That Can Speak and Listen","venue":"cs.CL","work_id":"a828d91a-9395-420e-825d-6858006f228e","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2306.12925","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:550fc0d2dcb2d9a0850ccd575bfafe678167b8a81734ef8368470727512f9ef6","observation_id":"1c982429-0c85-4830-9ecd-da24a5a83063","resolution":{"observed_at":"2026-05-16T21:51:17.782063Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.iwslt-1.2","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating","venue":null,"work_id":"3bd7446a-cf85-4274-b3e9-f8d818edab62","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:64d4a1b78dd4dcf5f6b7959a34fcd83313f2b41fc9a97a4e7d3264d74f63b210","observation_id":"7a4952b1-92ba-4d1a-8305-ba1eec819f97","resolution":{"observed_at":"2026-05-16T21:51:17.613097Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.09528","doi":"10.48550/arxiv.2510.09528","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":null,"venue":null,"work_id":"8177aa3c-b297-482b-ba21-774a76d6b36f","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:16f5ca624fb99a114018887b72eb9eb80a11aae12a0270d6de4eda112c59ec9f","observation_id":"a57069d7-24f7-4bac-9ac2-00dcae14dc54","resolution":{"observed_at":"2026-05-16T21:51:17.610601Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.acl-short.126","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"befebb78-4b29-485c-a0c9-3db211e0ce68","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:f10031c91341186e2c821838821027e7f0fdf454ce3ce4e84cb30160bc63fffd","observation_id":"e7311ac9-9369-4634-949c-bd0a47f224db","resolution":{"observed_at":"2026-05-16T21:51:17.544601Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2025.101257","doi":"10.1016/j.patter.2025.101257","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":"Patterns","work_id":"a69d6858-98be-4f22-a1b0-c29e2430efdc","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:172f8d917edfdc3fbac1e317b127bcfe20db6dc48e109042d1181e339c987707","observation_id":"b05066ef-ae83-4531-8c1f-1387657e4864","resolution":{"observed_at":"2026-05-16T21:51:17.550910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1145/3129340","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Schuller","venue":"Communications of the ACM","work_id":"0eb4340d-918e-4eda-aa93-499a1b909f45","year":2018},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:d976b1d2ebbbe67a370d19b677f023a4337ef1d83f7be4281732514d33142e4e","observation_id":"5f2614e4-8fc0-4b6c-b3f7-3acd1bf6521d","resolution":{"observed_at":"2026-05-16T21:51:17.609126Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"353c9eed-8224-471c-96f2-04cf1d2b0848","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:188e83b897fc72697499b5fda3c97c5a5d39b39ba4a55ca2c65e0c18aed0af30","observation_id":"2edd3dd5-b518-4379-a582-7b039a528899","resolution":{"observed_at":"2026-05-16T21:51:18.229691Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.14128","doi":"10.48550/arxiv.2509.14128","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2509.14128 , year =","venue":"ArXiv.org","work_id":"7cfdf406-6e6c-4f9d-8072-9e9796b4ced8","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:c0990fd0acbfd1dc244a4ac9fae72144e2ee30c6b07dbcbf0f7455bd1bd12226","observation_id":"7d0b7935-e4a0-4263-b8fc-1293416da29e","resolution":{"observed_at":"2026-05-16T21:51:17.805139Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07937","last_updated":"2024-12-09T13:43:15Z","snapshot_observed_at":"2026-08-07T03:56:59.593891Z","submitted_at":"2024-03-08T08:10:29Z","title":"Speech Robust Bench: A Robustness Benchmark For Speech Recognition","version":3},"cited_work":{"arxiv_id":"2403.07937","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.07937","snapshot_observed_at":"2026-06-29T11:13:20.727990Z","title":"Shah, David Solans Noguero, Mikko A","venue":null,"work_id":"673528ec-1520-4f66-a78f-4df7762c8262","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2403.07937","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:9fb84db364bd0b33a66a0915d8df92fb26a55e65aa3e9d62e99d6872da87e045","observation_id":"d7807f98-0e56-4b1b-95da-c294049409d6","resolution":{"observed_at":"2026-05-16T21:51:17.795119Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2002.05202","last_updated":"2020-02-12T19:57:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-02-12T19:57:13Z","title":"GLU Variants Improve Transformer","version":1},"cited_work":{"arxiv_id":"2002.05202","doi":"10.48550/arxiv.2002.05202","metadata_source":"pith","pith_arxiv_id":"2002.05202","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GLU Variants Improve Transformer","venue":"cs.LG","work_id":"17d0763c-1016-41ab-a478-478e890765eb","year":2020},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2002.05202","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:c8f1c18be2914cfaa8dd4b1c99e8818d64ce5495c5c75e8dabc2a09cd242a93e","observation_id":"fa676003-aadb-44f0-8959-08dbb68d7453","resolution":{"observed_at":"2026-05-16T21:51:17.797871Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-13T15:50:07.002485+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T15:50:07.002485+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1510.08484","last_updated":"2015-10-28T20:59:04Z","snapshot_observed_at":"2026-07-06T04:34:36.474437Z","submitted_at":"2015-10-28T20:59:04Z","title":"MUSAN: A Music, Speech, and Noise Corpus","version":1},"cited_work":{"arxiv_id":"1510.08484","doi":null,"metadata_source":"pith","pith_arxiv_id":"1510.08484","snapshot_observed_at":"2026-07-10T12:57:07.598373Z","title":"MUSAN: A Music, Speech, and Noise Corpus","venue":"cs.SD","work_id":"7c604702-578b-4f91-9cc6-f8aa7dbe6d26","year":2015},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/1510.08484","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:4c80d26456d4224602519920cd6a11764753669a00a4cc991bf78c5ea3dd55f4","observation_id":"de0be237-fc26-43d1-9624-10063a80bf08","resolution":{"observed_at":"2026-05-16T21:51:17.784630Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17797","last_updated":"2025-02-25T03:02:24Z","snapshot_observed_at":"2026-08-05T21:58:36.776468Z","submitted_at":"2025-02-25T03:02:24Z","title":"Enhancing Human Evaluation in Machine Translation with Comparative Judgment","version":1},"cited_work":{"arxiv_id":"2502.17797","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17797","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"22db18b2-6c20-46a7-9bdd-2d1d0b6dbd9f","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2502.17797","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:c81c7f573d327f8e2000ede827b34fcd55c64bd09b29b079749a5e6ab3b46c9a","observation_id":"cf06fe65-fd0f-4efd-8009-00efa5527418","resolution":{"observed_at":"2026-05-16T21:51:17.740476Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2020.acl-main.661","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"e6b38757-7058-48b1-a340-58e564749750","year":2020},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:7821df404d5d528a7a52912c374b89ded54ded9bf64843720cfb50dfbe1f4f98","observation_id":"0b707a6a-aadd-4f6a-9a7a-9e0882a3b73a","resolution":{"observed_at":"2026-05-16T21:51:17.600285Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/p19-1164","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Smith, and Luke Zettlemoyer","venue":null,"work_id":"967aa4f3-f461-43ac-8668-616d37c49a45","year":2019},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:c3729054e93e3938984f2580ac5f0ef76d9fb499c04cddc73d1804dec743add9","observation_id":"32493e8c-29f3-402f-a304-a8587f3f26ea","resolution":{"observed_at":"2026-05-16T21:51:17.526977Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.acl-long.336","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"In: Ku, L.-W., Martins, A., Srikumar, V","venue":null,"work_id":"2b6775f3-defe-49b7-98b1-323a9134954d","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:6902d5b4fec6e78fe2442292a020c7866719caf24e35672a0ff50540b9e0af7f","observation_id":"dc2b5175-a3f1-426b-9caf-c8a733255e54","resolution":{"observed_at":"2026-05-16T21:51:17.545032Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.09864","last_updated":"2023-11-08T13:36:32Z","snapshot_observed_at":"2026-07-06T11:01:58.137141Z","submitted_at":"2021-04-20T09:54:06Z","title":"RoFormer: Enhanced Transformer with Rotary Position Embedding","version":5},"cited_work":{"arxiv_id":"2104.09864","doi":"10.48550/arxiv.2104.09864","metadata_source":"pith","pith_arxiv_id":"2104.09864","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RoFormer: Enhanced Transformer with Rotary Position Embedding","venue":"cs.CL","work_id":"4e5eee26-cd04-4c7a-988f-3e6d1a1f0eb9","year":2021},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2104.09864","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:1069fcb20b0b5e9ef4cef9eb4478a799302bbfd6a8d33f9bf2a116f3af92ca6b","observation_id":"e0d9d294-1cd7-4658-b270-a6909818f95d","resolution":{"observed_at":"2026-05-16T21:51:17.779159Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:49:47.452101+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:49:47.452101+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23018","last_updated":"2025-06-25T09:38:19Z","snapshot_observed_at":"2026-07-06T21:32:38.137341Z","submitted_at":"2025-05-29T02:56:08Z","title":"EmotionTalk: An Interactive Chinese Multimodal Emotion Dataset With Rich Annotations","version":3},"cited_work":{"arxiv_id":"2505.23018","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.23018","snapshot_observed_at":"2026-06-29T13:53:29.130108Z","title":"InProceedings of the Seventh Conference on Machine Translation (WMT), pages 578–585, Abu Dhabi, United Arab Emirates (Hybrid)","venue":null,"work_id":"5b797a1e-3702-4598-ae54-d523e6b63811","year":2022},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2505.23018","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:61eef01e79c186993a04afb6ad6c9e065637e2e34f74d73855d0efe98332ca89","observation_id":"fd1a17c0-0349-4009-af21-07b894a831aa","resolution":{"observed_at":"2026-05-16T21:51:17.809163Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":"2302.13971","doi":"10.48550/arxiv.2302.13971","metadata_source":"pith","pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaMA: Open and Efficient Foundation Language Models","venue":"cs.CL","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:2cdd009357b00826cb749faddb5d4f45bb54a2056f6b8d59fa56e6b34adf05df","observation_id":"50cc316f-3b5f-400e-8312-1be3fa9b6bfb","resolution":{"observed_at":"2026-05-16T21:51:17.730050Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.findings-emnlp.892","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"b273a093-ffe3-4133-be50-4212344e2a6b","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:f5b64a26e06b9dc8bf5df8c2de88d13364660e6b63435388c6630f8ae19cdae2","observation_id":"f0478949-b088-40b5-b2b7-b5d95f061f7f","resolution":{"observed_at":"2026-05-16T21:51:17.604308Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.wmt-1.119","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Speech Is More than Words: Do Speech-to-Text Translation Systems Leverage Prosody?","venue":null,"work_id":"d46bfc36-31a6-4252-a2c3-a4777cb5890b","year":2024},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:bf41ffbc0b162a1f590305ec92eae50f6c66ac0ffd1111e9ba4c8aac6b40eda9","observation_id":"6b1d0e0c-3159-45de-b723-3459c46b59c2","resolution":{"observed_at":"2026-05-16T21:51:17.528303Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.10310","last_updated":"2020-10-24T06:07:01Z","snapshot_observed_at":"2026-08-01T17:44:35.425095Z","submitted_at":"2020-07-20T17:53:35Z","title":"CoVoST 2 and Massively Multilingual Speech-to-Text Translation","version":3},"cited_work":{"arxiv_id":"2007.10310","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.10310","snapshot_observed_at":"2026-07-04T20:30:07.211214Z","title":"Covost 2 and massively multilingual speech-to-text translation","venue":null,"work_id":"6c9bffaf-0bd1-4fa9-ac70-56e01e042f08","year":2007},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2007.10310","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:9b2c0d5f622b069a3801a1f937ef426aa1de901a65ec4badcb4a91d59273cc7a","observation_id":"f0f8e16e-a80c-4f4e-82a6-838529243c8b","resolution":{"observed_at":"2026-05-16T21:51:17.814383Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.iwslt-1.19","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"cf365065-eb1c-4a65-bc33-94e78293ccd3","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":101,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:f536e25df9458537b50ab16cd4dbb2f9d468a12dff20fe47305568e7c934465c","observation_id":"5b2432c0-ca59-44cd-9165-db966a0ee51f","resolution":{"observed_at":"2026-05-16T21:51:17.582139Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21437/interspeech.2018-1456","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ESPnet: End-to-end speech processing toolkit","venue":null,"work_id":"9e751e2f-538a-401e-9530-676a7e8d6aeb","year":2018},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":102,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:849ce170bdae7fde34058dd1276fc9e544d3b7e6f23f959852e2878de0635970","observation_id":"0c59d373-5c29-4fc3-b122-0aca5a225659","resolution":{"observed_at":"2026-05-16T21:51:17.595967Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21437/interspeech.2017-503","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Weiss, Jan Chorowski, Navdeep Jaitly, Yonghui Wu, and Zhifeng Chen","venue":null,"work_id":"44671455-efc7-437a-99ce-b95160498a4a","year":2017},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":103,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:72ddd7c4651116cf92643e2f5628e7399b9bc52683a74408f75ea2b01ca05766","observation_id":"6aff08d9-f8b0-418c-a0ab-2c7670a58e23","resolution":{"observed_at":"2026-05-16T21:51:17.584158Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.24963/ijcai.2023/761","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"1be40bdf-3187-4d83-9598-d9f90fa667dc","year":2023},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:4af100bc9b96cb5546b866be91c5219dd598304aaa9873253372a17ee62c8f67","observation_id":"cd8f56fc-2b79-41be-88ac-259f60d6c125","resolution":{"observed_at":"2026-05-16T21:51:17.614084Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21437/interspeech.2025-2247","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"812ea110-06c5-4790-a520-63b2ab4db563","year":2025},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:dcc442a19e2194a187b3e036735925d7aefe3eab09a858d9879485d6c906d797","observation_id":"57473c99-490c-4464-ab73-206420578892","resolution":{"observed_at":"2026-05-16T21:51:17.616231Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","latest_version":4,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-02T18:57:30.783881Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":1,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":13,"verified_exact":78,"verified_fuzzy":5},"total_outbound_references":106},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 100 of 106 outbound references and 6 inbound Pith citation observations for arXiv:2512.16378."}