{"as_of":"2026-08-11T07:16:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:cda6c7ce038e130d9455ba71de820ccfcc84c9aca6501c3ef6978a8227357a7c","coverage":[{"denominator":42,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T18:42:09.257514Z","state":"measured"},{"denominator":45,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":45,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T04:22:00.993596Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-05T16:07:18.631370Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.07025","snapshot_observed_at":"2026-08-02T18:42:04.589553Z","title":"Cascaded ASR-to- LLM pipelines convert speech signals to text early in the ASR stage, leading to information beyond semantics being discarded","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:04.589553Z"},"links":{"cited_paper":"/paper/2603.07025","citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:eef23b719aa818ad678141ae5ab0240df853a415d09329519a2f93b9f5f18905","observation_id":"2d776383-6d61-4e2a-801c-b4499603ce22","resolution":{"observed_at":"2026-08-02T18:42:04.589553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"cited_work":{"arxiv_id":"2603.07025","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.07025","snapshot_observed_at":"2026-08-05T16:07:18.631370Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","venue":"cs.CL","work_id":"0d10cbe0-eed3-4a99-85ea-d0a484f02bf9","year":2026},"citing_paper":{"arxiv_id":"2608.03610","last_updated":"2026-08-10T11:37:39Z","snapshot_observed_at":"2026-08-11T06:23:56.499422Z","submitted_at":"2026-08-04T13:02:47Z","title":"Language-Specialized Multi-Teacher On-Policy Distillation for Multilingual LLM-Based ASR","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T16:07:18.552775Z"},"links":{"cited_paper":"/paper/2603.07025","citing_paper":"/paper/2608.03610"},"observation_digest":"sha256:5e336c6e0fb6fb297907fe86ca4f47f04ef73bf6e3c554f7846dec57bafcfbf1","observation_id":"9e777564-50a2-41d4-9626-31351a41fe11","resolution":{"observed_at":"2026-08-05T16:07:18.637568Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.07025","snapshot_observed_at":"2026-08-11T04:22:00.993596Z","title":"Language-aware distillation for multilingual instruction-following speech llms with asr-only supervision,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.03610","last_updated":"2026-08-10T11:37:39Z","snapshot_observed_at":"2026-08-11T06:23:56.499422Z","submitted_at":"2026-08-04T13:02:47Z","title":"Language-Specialized Multi-Teacher On-Policy Distillation for Multilingual LLM-Based ASR","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T04:22:00.993596Z"},"links":{"cited_paper":"/paper/2603.07025","citing_paper":"/paper/2608.03610"},"observation_digest":"sha256:bbfb1e263f6e1b82ea4bbd554d134166b70e76d626b274612ed217cc21e923f9","observation_id":"5a76a0e6-462b-419f-a0a8-86fd3c19e535","resolution":{"observed_at":"2026-08-11T04:22:00.993596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2603.07025/citation-record","integrity":"/paper/2603.07025/integrity","json":"/paper/2603.07025/citation-record.json","paper":"/paper/2603.07025"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.07025","snapshot_observed_at":"2026-08-02T18:42:04.589553Z","title":"Cascaded ASR-to- LLM pipelines convert speech signals to text early in the ASR stage, leading to information beyond semantics being discarded","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:04.589553Z"},"links":{"cited_paper":"/paper/2603.07025","citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:eef23b719aa818ad678141ae5ab0240df853a415d09329519a2f93b9f5f18905","observation_id":"2d776383-6d61-4e2a-801c-b4499603ce22","resolution":{"observed_at":"2026-08-02T18:42:04.589553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:04.661254Z","title":"audio tail","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:04.661254Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:e2dc0f3a2b398e6bc79c4cafc9d1ea1645b9e40ec40350d4e5e2618f44fbfd3a","observation_id":"9a584b0f-fa38-4489-8c7c-ee692f3be648","resolution":{"observed_at":"2026-08-02T18:42:04.661254Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:04.743514Z","title":"Training Data Only annotated ASR data was used during the training of the model","venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:04.743514Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:4415d0e32040b2f6b893be5a3cd9fcd910f4981b49f4e612f381f34a08139bbb","observation_id":"44961a6c-26e1-42b0-8ec9-5c0f29d1ca5c","resolution":{"observed_at":"2026-08-02T18:42:04.743514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:04.806582Z","title":"answer not found","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:04.806582Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:f65bbcf680f2e526f645d05ce7fdf1d9872066a4455888a276e130353026dcd6","observation_id":"04599251-548c-4527-badc-18efa9c6f6d2","resolution":{"observed_at":"2026-08-02T18:42:04.806582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:04.895799Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:04.895799Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:c7af4b76401ca176534f000713a748df7a4b984aee932c866c254797e43ee430","observation_id":"6b83f37e-d1ee-4e41-b214-b836e734295d","resolution":{"observed_at":"2026-08-02T18:42:04.895799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:04.986371Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:04.986371Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:86d37aa1ac97baecdfe3a52c34e317d0bdc2f356538f727d3e932b00e037b20e","observation_id":"b280e5b3-31b5-457f-b7a2-88144e4f0aee","resolution":{"observed_at":"2026-08-02T18:42:04.986371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:05.083671Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:05.083671Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:fafa8f0bff365e21c1cc8dd80cb1cd0177a678eb368365bbfba63def956a9c45","observation_id":"c83efec9-dcd4-4b1c-8641-cc413fdf9725","resolution":{"observed_at":"2026-08-02T18:42:05.083671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:05.134538Z","title":"Blsp-emo: Towards empathetic large speech-language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:05.134538Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:3b254b5f46650b2686b95a25e0041c83b04e73b11689c1d1dcc7fac3229ad87f","observation_id":"4f29f94d-4d09-47e6-9346-e977d30fc379","resolution":{"observed_at":"2026-08-02T18:42:05.134538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:05.215941Z","title":"Benchmarking contextual and par- alinguistic reasoning in speech-llms: A case study with in-the- wild data,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:05.215941Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:45fa6438a0225d7281d5209c2b9e315adc644a9ccae1cf33e88294ed31314749","observation_id":"c8475616-154a-4090-816e-d98c27762d2e","resolution":{"observed_at":"2026-08-02T18:42:05.215941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:05.313762Z","title":"Generative spoken dialogue language modeling,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:05.313762Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:356cd75c8f15a8cc40a1d8173782fa11fc6b15de73c4e3789989c30d2bee8f1f","observation_id":"3091e165-bc04-46bf-9c86-8a104fde565c","resolution":{"observed_at":"2026-08-02T18:42:05.313762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07919","last_updated":"2023-12-21T10:20:42Z","snapshot_observed_at":"2026-08-07T10:17:55.688598Z","submitted_at":"2023-11-14T05:34:50Z","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07919","snapshot_observed_at":"2026-08-02T18:42:05.405658Z","title":"Qwen-audio: Advancing universal audio un- derstanding via unified large-scale audio-language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:05.405658Z"},"links":{"cited_paper":"/paper/2311.07919","citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:779c1ba7ee8911f281cf142ead4f693155e718c93a7b85a9ac614b27182ebdec","observation_id":"dec17be9-5a17-4f83-abe7-6e3b2f56bacb","resolution":{"observed_at":"2026-08-02T18:42:05.405658Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10759","last_updated":"2024-07-15T14:38:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T14:38:09Z","title":"Qwen2-Audio Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10759","snapshot_observed_at":"2026-08-02T18:42:05.496963Z","title":"Qwen2-audio technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:05.496963Z"},"links":{"cited_paper":"/paper/2407.10759","citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:e505184daf2c8d25101beb9eaa48f0ba3f254db3fd0af5ec242e2358f0be7f70","observation_id":"d61a9ffa-f886-4aae-81d8-8c2e9a2a68f3","resolution":{"observed_at":"2026-08-02T18:42:05.496963Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:05.586914Z","title":"Salmonn: Towards generic hearing abilities for large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:05.586914Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:4eeaed9b599976ded07e854a8286bc33ac31a0f5779be5137a9b5a7b4fe5d8e7","observation_id":"877b64c4-6e9c-4747-90f3-d15e321741bb","resolution":{"observed_at":"2026-08-02T18:42:05.586914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:05.651920Z","title":"Salm: Speech- augmented language model with in-context learning for speech recognition and translation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:05.651920Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:606b160e30cad3aa5352d4ada185a51f20debdf62fda222a73fc18137bdeb3b3","observation_id":"79c36c31-bada-40d1-9d98-95502e8724b7","resolution":{"observed_at":"2026-08-02T18:42:05.651920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:05.744796Z","title":"Meralion-audiollm: Advancing speech and language understanding for singapore,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:05.744796Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:dd6d73fa5811d2e9018cbdc0f823435bad8b3f59b6395782001d9ae81adf8954","observation_id":"f080d2c4-ffa0-4b34-8440-403ed56c69e2","resolution":{"observed_at":"2026-08-02T18:42:05.744796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:05.811297Z","title":"Seallms-audio: Large audio- language models for southeast asia,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:05.811297Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:9a6955b68b2aee94493dc8c562ea89c289e454ba0c2b6518db91116354da0d8c","observation_id":"9ec23560-1212-45bc-bf10-7dc7578092fe","resolution":{"observed_at":"2026-08-02T18:42:05.811297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.06666","last_updated":"2025-03-01T12:59:49Z","snapshot_observed_at":"2026-07-06T19:13:20.458958Z","submitted_at":"2024-09-10T17:34:34Z","title":"LLaMA-Omni: Seamless Speech Interaction with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.06666","snapshot_observed_at":"2026-08-02T18:42:05.886184Z","title":"Llama- omni: Seamless speech interaction with large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:05.886184Z"},"links":{"cited_paper":"/paper/2409.06666","citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:735e7827eac1261122c77c6498f46085abc6092218459a2b9e5ede6452b89d6e","observation_id":"941f052b-a146-45b8-b2c5-7596fdacdd7f","resolution":{"observed_at":"2026-08-02T18:42:05.886184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:05.949544Z","title":"Analyzing mitigation strategies for catastrophic forgetting in end-to-end training of spoken language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:05.949544Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:f2f9cf9b5f49cc01673a69623945435a9f9ad921e715822a5a2aabc4173ba1cc","observation_id":"891af285-33fc-45db-ae78-5ce85ee7e630","resolution":{"observed_at":"2026-08-02T18:42:05.949544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:06.015597Z","title":"Dis- tilling an end-to-end voice assistant without instruction training data,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:06.015597Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:55a5c1372431ce87e650c5568b65dbc0fb9f68b4785e3f5dae0fdf078e6df35d","observation_id":"b9910c52-fc4a-4ff8-b991-e31a3cc73d0c","resolution":{"observed_at":"2026-08-02T18:42:06.015597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:06.212757Z","title":"Tango 2: Aligning diffusion-based text- to-audio generations through direct preference optimization,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:06.212757Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:ac02d03824023e0b93ba668a157df904fddb4ff150291d01ac431e3ed35e5a07","observation_id":"c1c325b2-2d33-42b3-a3f0-2ffb39d09048","resolution":{"observed_at":"2026-08-02T18:42:06.212757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:06.390935Z","title":"Instruction data generation and unsupervised adap- tation for speech language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:06.390935Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:00a42f5b67670f275994159e86d9d652409c339c67e08bab9d48bef19f6e642c","observation_id":"3b3195cb-383a-4324-88da-30c2b9442ba6","resolution":{"observed_at":"2026-08-02T18:42:06.390935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:06.582277Z","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:06.582277Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:57eb0f33c2354ea27315655e0934e44715e6062e6f0cd5e372b690b19f2efbb3","observation_id":"effcaceb-e8ec-45c6-bcc9-e245ad1357c5","resolution":{"observed_at":"2026-08-02T18:42:06.582277Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.00916","last_updated":"2024-05-28T14:26:28Z","snapshot_observed_at":"2026-08-10T11:00:55.009445Z","submitted_at":"2023-09-02T11:46:05Z","title":"BLSP: Bootstrapping Language-Speech Pre-training via Behavior Alignment of Continuation Writing","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.00916","snapshot_observed_at":"2026-08-02T18:42:06.739946Z","title":"Blsp: Bootstrapping language-speech pre-training via behavior alignment of continuation writing,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:06.739946Z"},"links":{"cited_paper":"/paper/2309.00916","citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:94badcb8a464d19c577c8372293773b97a1050e034533a4641d32827a1515b07","observation_id":"29adec66-3801-409b-ae2b-d5541b2a59f6","resolution":{"observed_at":"2026-08-02T18:42:06.739946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:06.901189Z","title":"Inte- grating speech self-supervised learning models and large language models for asr,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:06.901189Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:8e775c3fb882ad30df95c95694323e45034af9887efeab783f5fa8c3136a6bb2","observation_id":"ff80b5db-251e-4481-a1c2-c17a27b628c2","resolution":{"observed_at":"2026-08-02T18:42:06.901189Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:07.105859Z","title":"An end-to-end speech summarization using large lan- guage model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:07.105859Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:6e2d5ac75cd468b27721474d32af042fb1d950e53b7403c11058e82d6ec0083e","observation_id":"6e6cc7a6-7984-411c-8c96-0aa38fa6115e","resolution":{"observed_at":"2026-08-02T18:42:07.105859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:07.244639Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:07.244639Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:b7aa90de2afdc0d745550ec13497db017ba847d7bb19f60b412b5aa5a41a2bd2","observation_id":"e873a736-7b59-4a06-9ac7-8631300c3c16","resolution":{"observed_at":"2026-08-02T18:42:07.244639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:07.426368Z","title":"Robust speech recognition via large-scale weak su- pervision,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:07.426368Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:4f99dc7d2c70ec430ca1885103b851cd0d114238e2375ff49fd20e4ad03e1033","observation_id":"d71c9ade-929a-41a9-88f1-011f16bc14ac","resolution":{"observed_at":"2026-08-02T18:42:07.426368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:07.648311Z","title":"A two-stage lora strategy for expanding language capabilities in multilingual asr models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:07.648311Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:5ea55ae22dfbc7a125bdcf44edfe9cb8d8cc6631d00ba329592acaea61d1ca9a","observation_id":"d7915a65-e7cc-4d66-a58b-b2f7dfe49549","resolution":{"observed_at":"2026-08-02T18:42:07.648311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:07.828697Z","title":"Language-routing mixture of experts for multilingual and code-switching speech recognition,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:07.828697Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:d0200860d7ebc1d6e16ae5a42e0cfba79a1a5dda0c894ada7d2127632dfaccec","observation_id":"58b8e062-eda9-4b4a-a554-ea6da2e90c30","resolution":{"observed_at":"2026-08-02T18:42:07.828697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:07.995689Z","title":"Adap- tive gating in mixture-of-experts based language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:07.995689Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:636f965e39e4e6881c07c23419e2668169090c867a8477a2910d18e6b43e09d6","observation_id":"64d1ae25-085d-4f4f-a907-118ed0d78ae2","resolution":{"observed_at":"2026-08-02T18:42:07.995689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:08.151807Z","title":"Sea-lion: Southeast asian languages in one network,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:08.151807Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:76ff8703fd1cb0ca915d621f092255380a78598fc26ce5621c9498109704176b","observation_id":"95e2c0d8-398b-4d26-b064-a8ebd9abb304","resolution":{"observed_at":"2026-08-02T18:42:08.151807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-10T16:40:37.411115Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-02T18:42:08.276139Z","title":"The llama 3 herd of models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:08.276139Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:a9e781dbca34e3563174f0140e5f9343ea94ef33f845ab7e6908314ca47fec3a","observation_id":"28299d65-a305-4da1-8af6-0c47b52e6c8b","resolution":{"observed_at":"2026-08-02T18:42:08.276139Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:08.383015Z","title":"Common voice: A massively-multilingual speech corpus,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:08.383015Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:f06e5e930e61ebb4526b0cce4dad437bd71a051de88c5c83eb44a89054d27c8f","observation_id":"96a16de5-4d96-4d9e-8bdb-cbe3d20fcd95","resolution":{"observed_at":"2026-08-02T18:42:08.383015Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:08.495564Z","title":"viV oice: Enabling vietnamese multi-speaker speech synthesis,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:08.495564Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:17e042e7f940dd5f9af410fd357bd97690b3fcd487b07c06e799957aea70336e","observation_id":"606e5f28-0483-461a-b7d7-61d18c51d90a","resolution":{"observed_at":"2026-08-02T18:42:08.495564Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:08.586485Z","title":"Yodas: Youtube-oriented dataset for audio and speech,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:08.586485Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:cdc79452273db933cffa0471a7473714b6467499546bc11fe602fb260ddefd2c","observation_id":"70a37499-afbb-4332-bff3-350dde3dfdb2","resolution":{"observed_at":"2026-08-02T18:42:08.586485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:08.671725Z","title":"Magic data open source cor- pus (id: 101),","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:08.671725Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:72bcf6aab6698ffb272f4909cd128298d10d1f5c09f774dddadd99ad8a383c69","observation_id":"d785b59c-9350-489e-b2c6-4b43dfbe66a7","resolution":{"observed_at":"2026-08-02T18:42:08.671725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:08.761675Z","title":"URO-bench: Towards comprehensive eval- uation for end-to-end spoken dialogue models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:08.761675Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:968c7099aeff620ba2530f3e8d95f010e06996fcb82305180148786609a02101","observation_id":"08956cfe-86b7-46c2-a69f-c5aacab12e00","resolution":{"observed_at":"2026-08-02T18:42:08.761675Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17196","last_updated":"2024-12-11T15:45:21Z","snapshot_observed_at":"2026-07-06T19:37:56.214143Z","submitted_at":"2024-10-22T17:15:20Z","title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17196","snapshot_observed_at":"2026-08-02T18:42:08.853789Z","title":"V oicebench: Benchmarking llm-based voice assistants,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:08.853789Z"},"links":{"cited_paper":"/paper/2410.17196","citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:9b3feee6f13e72e383e8fee2e8f961bea754dde86c96835b81c3dbd36e0de74e","observation_id":"9d6ab5e8-660d-4a5f-8848-bbcfebd5ab43","resolution":{"observed_at":"2026-08-02T18:42:08.853789Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:08.996908Z","title":"AudioBench: A universal benchmark for audio large language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:08.996908Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:bea225959b4ddcfbadb946df5319ba91939b387918e9f4479f603611c73b78f2","observation_id":"f5c89243-3be3-48c1-b189-ae7a610c04f0","resolution":{"observed_at":"2026-08-02T18:42:08.996908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.07475","last_updated":"2020-05-03T10:13:02Z","snapshot_observed_at":"2026-07-06T08:30:04.777443Z","submitted_at":"2019-10-16T17:05:21Z","title":"MLQA: Evaluating Cross-lingual Extractive Question Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.07475","snapshot_observed_at":"2026-08-02T18:42:09.093355Z","title":"Mlqa: Evaluating cross-lingual extractive question answering,","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:09.093355Z"},"links":{"cited_paper":"/paper/1910.07475","citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:10069a2e254405c92dd5ca0c1d718f1f1d2ffb4773350095384681c928f357d2","observation_id":"bf193e24-438a-425f-93ac-cb718708a220","resolution":{"observed_at":"2026-08-02T18:42:09.093355Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T18:42:09.197246Z","title":"PEDANTS: Cheap but effective and interpretable answer equiv- alence,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:09.197246Z"},"links":{"citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:6c7ea9c80fa3d5cc387b6a747a8c8bec2cfc1cf66ad90cd52247f661787ec211","observation_id":"d15af879-e5e4-4909-ab46-cdb6525757da","resolution":{"observed_at":"2026-08-02T18:42:09.197246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02612","last_updated":"2024-12-03T17:41:24Z","snapshot_observed_at":"2026-08-02T21:27:26.884251Z","submitted_at":"2024-12-03T17:41:24Z","title":"GLM-4-Voice: Towards Intelligent and Human-Like End-to-End Spoken Chatbot","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02612","snapshot_observed_at":"2026-08-02T18:42:09.257514Z","title":"Glm-4-voice: Towards intelligent and human-like end- to-end spoken chatbot,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-02T18:42:09.257514Z"},"links":{"cited_paper":"/paper/2412.02612","citing_paper":"/paper/2603.07025"},"observation_digest":"sha256:dce731fa596e08d2c56ac6463c11ddc0c4fd64769a1f1ebaf1a900ddd6bb0d07","observation_id":"24d8b381-7d8d-4761-8d9d-7a3e93f2bca4","resolution":{"observed_at":"2026-08-02T18:42:09.257514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2603.07025","last_updated":"2026-07-24T15:35:47Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-10T18:20:35.395998Z","submitted_at":"2026-03-07T04:09:47Z","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision"},"reference_resolution":{"displayed":42,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":42,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":42},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 42 of 42 outbound references and 3 inbound Pith citation observations for arXiv:2603.07025."}