{"as_of":"2026-08-19T13:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:34ed800f7eef97fbae0cc5849b6cafe48ced11f366fe071f58d93f1187de934b","coverage":[{"denominator":45,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:51:19.432830Z","state":"measured"},{"denominator":47,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":47,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:51:15.305445Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T11:58:44.561636Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.17417","snapshot_observed_at":"2026-08-07T14:51:15.305445Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:15.305445Z"},"links":{"cited_paper":"/paper/2505.17417","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:6f30ae1328087fa6f07da8bdd078e570afc7134c8f1f4dd70cd9cc107cd36d0f","observation_id":"11d56fbd-b22c-4a54-b72e-5a8c8ecf1c8e","resolution":{"observed_at":"2026-08-07T14:51:15.305445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"cited_work":{"arxiv_id":"2505.17417","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.17417","snapshot_observed_at":"2026-08-07T11:58:44.561636Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","venue":"eess.AS","work_id":"fb047d49-2daa-4f8f-83a6-13e06715a80c","year":2025},"citing_paper":{"arxiv_id":"2506.06343","last_updated":"2025-06-01T09:27:55Z","snapshot_observed_at":"2026-08-17T08:11:21.214669Z","submitted_at":"2025-06-01T09:27:55Z","title":"TESU-LLM: Training Speech-LLMs Without Speech via Unified Encoder Alignment","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:58:41.531921Z"},"links":{"cited_paper":"/paper/2505.17417","citing_paper":"/paper/2506.06343"},"observation_digest":"sha256:ce6b0749362daad05a4c51ade4a7e49a78141a501551d15ec8e65ca1599cf5a5","observation_id":"6df42d5f-d314-4609-a221-5560619ae578","resolution":{"observed_at":"2026-08-07T11:58:44.627131Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.17417/citation-record","integrity":"/paper/2505.17417/integrity","json":"/paper/2505.17417/citation-record.json","paper":"/paper/2505.17417"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.17417","snapshot_observed_at":"2026-08-07T14:51:15.305445Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:15.305445Z"},"links":{"cited_paper":"/paper/2505.17417","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:6f30ae1328087fa6f07da8bdd078e570afc7134c8f1f4dd70cd9cc107cd36d0f","observation_id":"11d56fbd-b22c-4a54-b72e-5a8c8ecf1c8e","resolution":{"observed_at":"2026-08-07T14:51:15.305445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:23.642386Z","title":"First, we train a residual vector quantizer (RVQ) to en- code speech into discrete semantic tokens that align with Whis- per’s encoder representations","venue":null,"work_id":"e67fdbe5-ea8b-4247-8887-951eed2ed1d3","year":null},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:15.341648Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:47d9fb4115f51eaf8198f36dc7efef5b2e620e5b1a4f504105bef37ce833161b","observation_id":"900b0269-1f87-4455-87c5-55626e2fffb6","resolution":{"observed_at":"2026-08-07T14:51:23.699872Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:23.478383Z","title":"Datasets For Stage 1, we utilized two automatic speech recognition (ASR) datasets: viV oice (Vietnamese) and LibriTTS-R[22] (English)","venue":null,"work_id":"a22d7321-0be6-4616-8dec-62a06bcc84f4","year":null},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:15.413140Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:204326034ff26bc3757156825d3783d29a143da558b61f4a020adaf22df0520c","observation_id":"51b3eb90-d3d4-4fbb-ae2d-e431f0cab151","resolution":{"observed_at":"2026-08-07T14:51:23.550068Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:23.243275Z","title":"ASR and Speechless Comparisons To evaluate the performance of the Speechless model alone, we make use of ASR test sets","venue":null,"work_id":"df185e9a-24bd-4638-bac0-d7c25c80f517","year":null},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:15.506491Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:bab475bab4bb7f9493a70f96fbf5cbde7e90954302bb2dbbbc35bad9f9993643","observation_id":"1702fe69-67d2-46e0-8a75-5b85d782328b","resolution":{"observed_at":"2026-08-07T14:51:23.379474Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:22.849823Z","title":"By lever- aging a quantized Whisper encoder, Speechless generates se- mantic speech tokens, effectively addressing challenges in low- resource languages","venue":null,"work_id":"2510e206-564b-44c8-af9c-e0415ee90d6c","year":null},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:15.761479Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:0110a9d41f219f8f8e849bf70cb3d423bd7278e745646226c4314432182d9a62","observation_id":"dacab840-508a-49a8-a599-669601336807","resolution":{"observed_at":"2026-08-07T14:51:22.925666Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:22.988303Z","title":"This is also clear when see that with added noise (VBD noisy), the Whisper encoder starts to generate tokens that show poorer WER in comparison","venue":null,"work_id":"cdcc33e0-f8ca-49ac-9c23-2bd541cdb2f0","year":null},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:15.667264Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:fe6a86b18804c2b4527c13297b2729e36cb47087d5eef2e3c8df07baa0f0568a","observation_id":"2c1e8d43-313a-4cff-8756-24ca52554909","resolution":{"observed_at":"2026-08-07T14:51:23.064274Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:16.343843Z","title":"SALMONN: Towards generic hearing abilities for large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:16.343843Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:8f7a6db7b1250a148db901c696ac5571717d5a760ffe879abb0f6dbfc8a91119","observation_id":"eaf8e15b-238e-44e0-a5c9-eadd4d06e17a","resolution":{"observed_at":"2026-08-07T14:51:16.343843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.15316","last_updated":"2025-04-04T08:29:19Z","snapshot_observed_at":"2026-08-18T09:58:21.178258Z","submitted_at":"2024-10-20T07:03:49Z","title":"Ichigo: Mixed-Modal Early-Fusion Realtime Voice Assistant","version":3},"cited_work":{"arxiv_id":"2410.15316","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.15316","snapshot_observed_at":"2026-08-07T14:51:19.701569Z","title":"Ichigo: Mixed-Modal Early-Fusion Realtime Voice Assistant","venue":"cs.CL","work_id":"2962cc4e-cfd1-41fa-8217-e0f0c99da4cb","year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:15.830544Z"},"links":{"cited_paper":"/paper/2410.15316","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:5556fd104e8624c1abee38faccded2b8c5835891c17869b0d71b6995620a6966","observation_id":"2345f4fd-92d2-4833-8314-f1a8d9e0a7c2","resolution":{"observed_at":"2026-08-07T14:51:19.788559Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13577","last_updated":"2024-11-26T09:20:48Z","snapshot_observed_at":"2026-08-15T08:49:03.887183Z","submitted_at":"2024-11-15T04:16:45Z","title":"WavChat: A Survey of Spoken Dialogue Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.13577","snapshot_observed_at":"2026-08-07T14:51:15.917688Z","title":"Wavchat: A survey of spoken dialogue models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:15.917688Z"},"links":{"cited_paper":"/paper/2411.13577","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:52ecade508fea6395b27c8df91c19b90c43ce62ad03ec858602d970f770c8bae","observation_id":"24297cbb-faa3-4c2a-a25a-9cf84fe8d9b4","resolution":{"observed_at":"2026-08-07T14:51:15.917688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03751","last_updated":"2025-08-07T13:29:43Z","snapshot_observed_at":"2026-08-16T13:13:34.592774Z","submitted_at":"2024-10-01T21:48:12Z","title":"Recent Advances in Speech Language Models: A Survey","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03751","snapshot_observed_at":"2026-08-07T14:51:15.984529Z","title":"Recent advances in speech language models: A survey,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:15.984529Z"},"links":{"cited_paper":"/paper/2410.03751","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:1611c94d9334a59bb27443894d516ac0f2d611b1d318e1ec1761e73a1aa30c61","observation_id":"8b62c3c8-bdc0-46e4-bea0-2e93f641eeb5","resolution":{"observed_at":"2026-08-07T14:51:15.984529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.06666","last_updated":"2025-03-01T12:59:49Z","snapshot_observed_at":"2026-08-16T13:19:40.907971Z","submitted_at":"2024-09-10T17:34:34Z","title":"LLaMA-Omni: Seamless Speech Interaction with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.06666","snapshot_observed_at":"2026-08-07T14:51:16.069890Z","title":"Llama- omni: Seamless speech interaction with large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:16.069890Z"},"links":{"cited_paper":"/paper/2409.06666","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:266347eda2b1a9d03be6ac7e96ec03e634589e8b7cc869a1a446c5e6cec6abcd","observation_id":"0d13692c-21c8-42a9-bc12-cfe3f818b5c3","resolution":{"observed_at":"2026-08-07T14:51:16.069890Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:22.662059Z","title":"Instruction data generation and unsupervised adap- tation for speech language models,","venue":null,"work_id":"31ada050-8b24-4274-9399-efa74f72d2ea","year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:16.167458Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:6a4987ccdb3fc157391a9655bd00a0514c6d199121b313023c9e2e87d2174053","observation_id":"a563852e-78bc-42c8-8b2d-111ec6d27c1c","resolution":{"observed_at":"2026-08-07T14:51:22.786323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:22.510735Z","title":"Tango 2: Aligning diffusion-based text-to-audio generations through direct preference optimization,","venue":null,"work_id":"fe361032-cc91-47cf-8f61-0aabe7b48d27","year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:16.269017Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:3deef64ee59423c5a7360b78737720b31b4bf4d0904b4c789abd97a42aeb2528","observation_id":"4f491142-bab0-4c43-95c5-2aa0aaed3b49","resolution":{"observed_at":"2026-08-07T14:51:22.581169Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02678","last_updated":"2024-10-03T17:04:48Z","snapshot_observed_at":"2026-08-19T07:46:42.192670Z","submitted_at":"2024-10-03T17:04:48Z","title":"Distilling an End-to-End Voice Assistant Without Instruction Training Data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02678","snapshot_observed_at":"2026-08-07T14:51:17.014767Z","title":"Dis- tilling an end-to-end voice assistant without instruction training data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:17.014767Z"},"links":{"cited_paper":"/paper/2410.02678","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:c637f419afea53bad183a4d293e31f5bafb9999008503bfe77ca762556ca0d65","observation_id":"caa0ee6b-f3a4-45e4-98b1-de8f6327de53","resolution":{"observed_at":"2026-08-07T14:51:17.014767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.02248","last_updated":"2024-06-14T17:57:13Z","snapshot_observed_at":"2026-08-16T14:46:05.128165Z","submitted_at":"2023-11-03T21:47:03Z","title":"COSMIC: Data Efficient Instruction-tuning For Speech In-Context Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.02248","snapshot_observed_at":"2026-08-07T14:51:16.436586Z","title":"Cosmic: Data efficient instruction-tuning for speech in-context learning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:16.436586Z"},"links":{"cited_paper":"/paper/2311.02248","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:7c41405092045fc979333fe333e2189f6e0178c27af22df535853b18d50bbaa5","observation_id":"607d190d-46da-4014-af69-c27e2ebce769","resolution":{"observed_at":"2026-08-07T14:51:16.436586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.10390","last_updated":"2024-04-18T08:13:58Z","snapshot_observed_at":"2026-08-16T15:07:09.479698Z","submitted_at":"2023-08-20T23:47:23Z","title":"LibriSQA: A Novel Dataset and Framework for Spoken Question Answering with Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.10390","snapshot_observed_at":"2026-08-07T14:51:16.525195Z","title":"Lib- risqa: Pioneering free-form and open-ended spoken question an- swering with a novel dataset and framework,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:16.525195Z"},"links":{"cited_paper":"/paper/2308.10390","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:6b063276e46c95ed1a752f6a3bf0f0d67049db59b1ccb0175a0c28c3a0bf7ded","observation_id":"a6c1edec-a87e-4e7f-a463-d1e943a859ce","resolution":{"observed_at":"2026-08-07T14:51:16.525195Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:22.351479Z","title":"An efficient and high fidelity vietnamese streaming end-to-end speech synthesis,","venue":null,"work_id":"1a21919b-c26f-400f-a7e2-8690f22c4e2e","year":2022},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:16.609766Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:ad8be29e30463049362386a935aa434feef6a5e4a475164e226d6455eb4fc64a","observation_id":"2c331902-1aca-4b33-9d3c-e062d51eaabd","resolution":{"observed_at":"2026-08-07T14:51:22.428889Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:22.176309Z","title":"Low-resource multilingual and zero-shot multispeaker TTS,","venue":null,"work_id":"05dd62e4-ba8e-421c-bc2f-72967061f97a","year":2022},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:16.720043Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:3466bceac809285e7aafcde6d722cb075d3ff7e03eb807ee16bb952ed141f1eb","observation_id":"53ecc50b-f8f7-484c-ba70-7a61eb396779","resolution":{"observed_at":"2026-08-07T14:51:22.262103Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.10999","last_updated":"2025-05-23T10:37:01Z","snapshot_observed_at":"2026-08-16T13:17:54.428873Z","submitted_at":"2024-09-17T09:04:03Z","title":"Enhancing Low-Resource Language and Instruction Following Capabilities of Audio Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.10999","snapshot_observed_at":"2026-08-07T14:51:16.797400Z","title":"Enhancing low-resource language and in- struction following capabilities of audio language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:16.797400Z"},"links":{"cited_paper":"/paper/2409.10999","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:cdd232ca2b48da406b968e60260e92b78625c1f3ec351d8a5c7e2f506fe13006","observation_id":"f546cba8-f8c1-4c1a-b2b0-7d4610d039f5","resolution":{"observed_at":"2026-08-07T14:51:16.797400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:22.008905Z","title":"Unsupervised cross-modal alignment of speech and text embedding spaces,","venue":null,"work_id":"5b677d27-5fb3-4809-af32-9f2fb735194e","year":2018},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:16.900732Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:6df8e03db5daef1aa315b1effd62d92b01f928e48b1bee146227fcc3ea72aea5","observation_id":"c790d769-9dcd-4498-b404-bb479547d23d","resolution":{"observed_at":"2026-08-07T14:51:22.078323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:20.989460Z","title":"Alpaca: A strong, replicable instruction-following model,","venue":null,"work_id":"ff626327-abab-4d9a-98c0-bfd1bbebceca","year":2023},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:17.763096Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:9956230195b911ee416f7e163e385fa9352a43574a8f8ac1ade92d8d43c8d567","observation_id":"a31e3807-8636-42a3-b356-ff6020e822fb","resolution":{"observed_at":"2026-08-07T14:51:21.053822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:21.791913Z","title":"An analysis of semantically-aligned speech-text embeddings,","venue":null,"work_id":"a54b3e6c-7433-4749-869d-8072ec520394","year":2023},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:17.124483Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:ab8c8f3717fce86e962531770944fbe44a213669db8103b30918bfb44372cbdd","observation_id":"7c2dd311-c4b4-4b2c-8cdf-0d8e87ddef20","resolution":{"observed_at":"2026-08-07T14:51:21.907697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:21.579890Z","title":"Astra: Aligning speech and text representa- tions for asr without sampling,","venue":null,"work_id":"f81da64a-3040-44d3-9de8-c9ed5445146f","year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:17.245173Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:690f1fa9960ede66c6af080520d9ad091a167aae8bed613b4d60d7102329ea19","observation_id":"c1591c75-ab13-4a3b-b6fc-2972fd1bf797","resolution":{"observed_at":"2026-08-07T14:51:21.694139Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:17.321338Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:17.321338Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:9ebc0c1f87283cd42929d2507cfee7ab5bb80554a8b6e2b49e2ec324068a1347","observation_id":"abf51dc9-7f4b-4e10-8e0f-db41f1142271","resolution":{"observed_at":"2026-08-07T14:51:17.321338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:21.420117Z","title":"Speecht5: Unified-modal encoder-decoder pre-training for spoken language processing,","venue":null,"work_id":"8de4328e-cec2-4e12-ae0f-e6bfb8b7e3d8","year":2021},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:17.427170Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:4abefbce88ab2ef3d432b8167c3a44135ddea902f43c2009cda79e0d53c57651","observation_id":"9347ce0c-ce5b-4d54-aaac-79a2f5c4f7cb","resolution":{"observed_at":"2026-08-07T14:51:21.495300Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:21.262614Z","title":"Sailor 2 dataset,","venue":null,"work_id":"9334c016-3e64-4912-afde-e7af3a909e8d","year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:17.534692Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:8241774a342cb8c87aaf469fc9a74bd90e78db94683ea4998ee530a32035e704","observation_id":"d1757e7c-322e-43a2-914b-f4543644f43c","resolution":{"observed_at":"2026-08-07T14:51:21.318498Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:21.135624Z","title":"Vtsnlp instruct general dataset,","venue":null,"work_id":"3e97b4f2-355f-4417-989e-eaf735a8dd15","year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:17.650905Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:9c39fb661c0c55211939fc691a802f71c71c0a2b69f8d67e2e78941d8ec79bed","observation_id":"d3fad88d-c6b6-4f61-954b-221df6fbe0e2","resolution":{"observed_at":"2026-08-07T14:51:21.209910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17196","last_updated":"2024-12-11T15:45:21Z","snapshot_observed_at":"2026-08-17T12:45:25.032324Z","submitted_at":"2024-10-22T17:15:20Z","title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17196","snapshot_observed_at":"2026-08-07T14:51:18.448509Z","title":"V oicebench: Benchmarking llm-based voice assistants,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:18.448509Z"},"links":{"cited_paper":"/paper/2410.17196","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:2866275fef647ce3eed1e9e99333c269ff80cea01b07fff635a0165cea578a06","observation_id":"a89f7a66-63ba-40b4-9f0a-949c3ae289c1","resolution":{"observed_at":"2026-08-07T14:51:18.448509Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.02882","last_updated":"2019-04-05T06:05:00Z","snapshot_observed_at":"2026-08-15T16:44:02.859541Z","submitted_at":"2019-04-05T06:05:00Z","title":"LibriTTS: A Corpus Derived from LibriSpeech for Text-to-Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.02882","snapshot_observed_at":"2026-08-07T14:51:17.873469Z","title":"Libritts: A corpus derived from librispeech for text- to-speech,","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:17.873469Z"},"links":{"cited_paper":"/paper/1904.02882","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:04a50c14f124f3b5a5114323ae14a1ed69615fafbfe90153d794ec46f65c1a15","observation_id":"9cfdabc0-e0dc-4eb3-8c5c-aa231bd844b1","resolution":{"observed_at":"2026-08-07T14:51:17.873469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:20.806940Z","title":"vivoice: Enabling vietnamese multi-speaker speech synthesis,","venue":null,"work_id":"bbaf8bc1-a4ce-4461-9aec-9f3a840d5bcb","year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:17.955220Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:a884a6e350d9f5c6bee55604f352bc359d3db9ef353acb9592da4926bfafe029","observation_id":"acede9d8-bcef-4531-b951-8ca6819b8fb1","resolution":{"observed_at":"2026-08-07T14:51:20.884120Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:23.122628Z","title":null,"venue":null,"work_id":"7c0adc37-274d-4af8-8384-26f2f0af9fd0","year":null},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:15.602667Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:3e360a213fac4a940f04506722879472616cf91cddff0df744491e3d99771858","observation_id":"389b06a8-4a30-4540-af90-b95b5f135e3a","resolution":{"observed_at":"2026-08-07T14:51:23.175121Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:20.655791Z","title":"Libritts-p: A corpus with speaking style and speaker identity prompts for text-to-speech and style captioning,","venue":null,"work_id":"c7e57f6b-975f-4b04-a271-ab73fb24547f","year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:18.061792Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:2c12a6f364ab83a76d767401073bafa756c1ea1e9050884b7b38457f10262cd1","observation_id":"0cc6ac39-bfac-41f9-91ac-54f907faddbf","resolution":{"observed_at":"2026-08-07T14:51:20.728142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.03411","last_updated":"2020-12-19T09:18:21Z","snapshot_observed_at":"2026-08-15T16:45:09.638988Z","submitted_at":"2020-12-07T01:53:45Z","title":"MLS: A Large-Scale Multilingual Dataset for Speech Research","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.03411","snapshot_observed_at":"2026-08-07T14:51:18.155649Z","title":"Mls: A large-scale multilingual dataset for speech research,","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:18.155649Z"},"links":{"cited_paper":"/paper/2012.03411","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:2afb0e3fd4c938dc6d4546e2538642a2243b91615e090fc343c0d061aa9a9f98","observation_id":"50bd9f11-e106-401b-9dec-ce817a39b5dd","resolution":{"observed_at":"2026-08-07T14:51:18.155649Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06180","last_updated":"2023-09-12T12:50:04Z","snapshot_observed_at":"2026-08-02T09:51:08.145755Z","submitted_at":"2023-09-12T12:50:04Z","title":"Efficient Memory Management for Large Language Model Serving with PagedAttention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06180","snapshot_observed_at":"2026-08-07T14:51:18.265223Z","title":"Efficient memory man- agement for large language model serving with pagedattention,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:18.265223Z"},"links":{"cited_paper":"/paper/2309.06180","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:1fe15b1cb2390d05ec5b9c866b7ec64f916d30640475d6d8b0e092efead3ab64","observation_id":"0e5160ce-8d1b-4afb-82b0-c0bd78df15ff","resolution":{"observed_at":"2026-08-07T14:51:18.265223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.05889","last_updated":"2018-09-30T03:14:16Z","snapshot_observed_at":"2026-08-15T15:55:11.242092Z","submitted_at":"2017-12-16T01:29:49Z","title":"Ray: A Distributed Framework for Emerging AI Applications","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.05889","snapshot_observed_at":"2026-08-07T14:51:18.378750Z","title":"Ray: A distributed framework for emerging ai applications,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:18.378750Z"},"links":{"cited_paper":"/paper/1712.05889","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:23e9540cbb8d36602f1de4e9780abb5951f91b01a517d65abd468808e69d4169","observation_id":"2b6be35c-d36b-4d4c-a7a6-61f9e11e547c","resolution":{"observed_at":"2026-08-07T14:51:18.378750Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:18.542975Z","title":"Lib- rispeech: an asr corpus based on public domain audio books,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:18.542975Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:ef5e250810eb6da383e2109ee770125f2afd1d96f3298411d9a310d3e377fadb","observation_id":"ee710545-1be9-43f2-8cd7-d8b722303ef4","resolution":{"observed_at":"2026-08-07T14:51:18.542975Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:20.490996Z","title":"Speech enhancement for a noise-robust text-to-speech synthe- sis system using deep recurrent neural networks,","venue":null,"work_id":"e7a1811c-505a-4f35-9a72-4bb51e69cd6d","year":2016},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:18.623468Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:d21f5dca794336aa0c019e90a3c01161cd6c0a5be9f6cdf0ffacfac7c5839670","observation_id":"abf34e67-8708-42fb-bb91-58d90c51c0d2","resolution":{"observed_at":"2026-08-07T14:51:20.561260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:18.760225Z","title":"The diverse environments multi-channel acoustic noise database (demand): A database of multichannel environmental noise recordings,","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:18.760225Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:6cc72cddbffa6cf49e3045a55dc28ecf35af12940601dbe49e5e67a8a4035c7a","observation_id":"18225890-ce84-465f-95d6-18d6f01ce73a","resolution":{"observed_at":"2026-08-07T14:51:18.760225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:20.316555Z","title":"Com- mon voice: A massively-multilingual speech corpus,","venue":null,"work_id":"c86c84a8-7f63-4655-98fe-952283dc2fc9","year":2020},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:18.825142Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:8beddb36b984b55d0f65ea68493380674cae51d0f57d5df5ac9b2dfa03ae9623","observation_id":"be6b2239-8b98-4800-b2aa-c7d0d53ef519","resolution":{"observed_at":"2026-08-07T14:51:20.395648Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.04475","last_updated":"2025-03-10T09:27:03Z","snapshot_observed_at":"2026-07-06T17:56:23.317089Z","submitted_at":"2024-04-06T02:29:02Z","title":"Length-Controlled AlpacaEval: A Simple Way to Debias Automatic Evaluators","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.04475","snapshot_observed_at":"2026-08-07T14:51:18.912898Z","title":"Length- controlled alpacaeval: A simple way to debias automatic evalua- tors,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:18.912898Z"},"links":{"cited_paper":"/paper/2404.04475","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:a0e7beb1fca30082b3ce8ecaaba58d0ae7f87ff3f4e5c69f7f70158401bd2597","observation_id":"520c3d00-e442-4328-a5f0-0dd46f1321b7","resolution":{"observed_at":"2026-08-07T14:51:18.912898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:20.124656Z","title":"SD-QA: Spoken dialectal question answering for the real world,","venue":null,"work_id":"80e6d706-b27c-415a-92fb-2dc30dec7987","year":2021},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:19.032030Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:cde52446ff6acdd78bdc765268867172f86fd4e1207a1d7f8439c2a29a806530","observation_id":"ccb497d4-3c47-4829-a70a-ba964cd1bfd5","resolution":{"observed_at":"2026-08-07T14:51:20.223654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.02789","last_updated":"2018-09-08T11:47:16Z","snapshot_observed_at":"2026-08-14T07:00:02.529732Z","submitted_at":"2018-09-08T11:47:16Z","title":"Can a Suit of Armor Conduct Electricity? A New Dataset for Open Book Question Answering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.02789","snapshot_observed_at":"2026-08-07T14:51:19.139471Z","title":"Can a suit of armor conduct electricity? a new dataset for open book question answering,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:19.139471Z"},"links":{"cited_paper":"/paper/1809.02789","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:05fc4ca7dec484e1ba6acfc50eddfcd4d46f66dc887a7d054a637ee0dbd16173","observation_id":"41d8c8c1-34f7-4e8a-ba81-3a2af63b33d9","resolution":{"observed_at":"2026-08-07T14:51:19.139471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15043","last_updated":"2023-12-20T20:48:57Z","snapshot_observed_at":"2026-08-12T09:06:50.363435Z","submitted_at":"2023-07-27T17:49:12Z","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15043","snapshot_observed_at":"2026-08-07T14:51:19.219916Z","title":"Universal and transferable adversarial attacks on aligned language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:19.219916Z"},"links":{"cited_paper":"/paper/2307.15043","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:7b52ded80f876a6e14d111585c241b7a87aedf02e204759f89be8b9da6e56553","observation_id":"9ccdc6d9-0ca4-42ca-8e96-0d8ac76004b9","resolution":{"observed_at":"2026-08-07T14:51:19.219916Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00037","last_updated":"2024-10-02T09:11:45Z","snapshot_observed_at":"2026-07-30T10:21:14.474746Z","submitted_at":"2024-09-17T17:55:39Z","title":"Moshi: a speech-text foundation model for real-time dialogue","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00037","snapshot_observed_at":"2026-08-07T14:51:19.350579Z","title":"Moshi: a speech-text foundation model for real-time dialogue,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:19.350579Z"},"links":{"cited_paper":"/paper/2410.00037","citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:cd436f12bb6f5e56c4ee0223aa24a96800af82dfb0c81729246e77e75e8389ef","observation_id":"bb07a2e8-90bc-4f87-a40a-8a4236a11a76","resolution":{"observed_at":"2026-08-07T14:51:19.350579Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:51:19.906451Z","title":"BLSP: Bootstrapping language-speech pre-training via behavior alignment of continuation writing,","venue":null,"work_id":"3d880810-34df-4975-a70f-42c73ff23ac9","year":2024},"citing_paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:19.432830Z"},"links":{"citing_paper":"/paper/2505.17417"},"observation_digest":"sha256:fbc53adda45d18f1bce53f81789acb3651bf2a2c963ca7f0f449780130ffa9a4","observation_id":"21615463-daac-4d2e-8f38-928b36f96ce0","resolution":{"observed_at":"2026-08-07T14:51:20.026378Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.17417","last_updated":"2025-05-23T03:05:47Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-17T02:24:20.568852Z","submitted_at":"2025-05-23T03:05:47Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages"},"reference_resolution":{"displayed":45,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":22,"verified_exact":1,"verified_fuzzy":22},"total_outbound_references":45},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 45 of 45 outbound references and 2 inbound Pith citation observations for arXiv:2505.17417."}