{"as_of":"2026-08-21T17:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e466f9556415fdbe119c8702e0ccfbe3b6aac78260e184aa9773f83576128707","coverage":[{"denominator":52,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":52,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-10T20:30:11.127007Z","state":"measured"},{"denominator":52,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":52,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.06827/citation-record","integrity":"/paper/2607.06827/integrity","json":"/paper/2607.06827/citation-record.json","paper":"/paper/2607.06827"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2304.03277","last_updated":"2023-04-06T17:58:09Z","snapshot_observed_at":"2026-08-16T13:26:40.822276Z","submitted_at":"2023-04-06T17:58:09Z","title":"Instruction Tuning with GPT-4","version":1},"cited_work":{"arxiv_id":"2304.03277","doi":"10.48550/arxiv.2304.03277","metadata_source":"pith","pith_arxiv_id":"2304.03277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Instruction Tuning with GPT-4","venue":"cs.CL","work_id":"fd515477-f9f1-48aa-9feb-a3308e7656bb","year":2023},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2304.03277","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:f4bf3be0ffcea20f3a22952507458310e5aaa1b29664fa9bd237ef6418fe694a","observation_id":"cf29194d-1cfa-489f-9ed6-5e7190f31635","resolution":{"observed_at":"2026-07-10T20:37:34.449922Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":"2407.21783","doi":"10.1016/s0749-0720(15","metadata_source":"pith","pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"The Llama 3 Herd of Models","venue":"cs.AI","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:d46e84e796120a4293a72a2be8ef24890813c8ba43746563b0dbfa74a948bef8","observation_id":"eea683b8-7027-4e02-bdaf-74fad92d9296","resolution":{"observed_at":"2026-07-10T20:37:34.423879Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:bdb89421786713eeea84a5715d1ed061b40c3fcabdfaf75a5d1bacf7728a6928","observation_id":"c43ec401-fa32-4f5c-a5b2-44a87e99bc8d","resolution":{"observed_at":"2026-07-10T20:37:34.409134Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.08905","last_updated":"2024-12-12T03:37:41Z","snapshot_observed_at":"2026-08-20T14:44:18.211453Z","submitted_at":"2024-12-12T03:37:41Z","title":"Phi-4 Technical Report","version":1},"cited_work":{"arxiv_id":"2412.08905","doi":"10.48550/arxiv.2412.08905","metadata_source":"pith","pith_arxiv_id":"2412.08905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Phi-4 Technical Report","venue":"cs.CL","work_id":"b6274271-7af9-4ee8-993b-ba1ba4205ba8","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2412.08905","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:2cc8ee01df6955f1d855e1677d81c9d9070fd9c13e7db2ad98883ef200531170","observation_id":"0d59d235-b32b-4ae5-8cbf-67d63b774509","resolution":{"observed_at":"2026-07-10T20:37:34.420712Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-02T03:08:11.323957+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T03:08:11.323957+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.712559Z","title":"SALMONN: Towards generic hearing abilities for large language models,","venue":null,"work_id":"57783a2a-c56b-428e-99e7-735fee684ac9","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:a8cd5e0ad9e8ffa7730c200a11d9ef6cff6b47ccee195ba2f6db67e76557f65f","observation_id":"bd8f82fb-f82f-4345-8fae-52185feaca23","resolution":{"observed_at":"2026-07-10T20:37:34.713988Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-08-17T18:50:07.059564Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":"2412.15115","doi":"10.1145/3581783.3612503","metadata_source":"pith","pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5 Technical Report","venue":"cs.CL","work_id":"d8432992-4980-4a81-85c7-9fa2c2b87f85","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:9f1a8e6231da8609db9f6c9bc2bcb169391202cbca6de647552bdd16fc7f80b9","observation_id":"0eb31cc9-519b-4462-a17f-166f730a67b2","resolution":{"observed_at":"2026-07-10T20:37:34.438595Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.710129Z","title":"Alignformer: Modality matching can achieve better zero-shot instruction-following speech-llm,","venue":null,"work_id":"63de6b50-f96d-4f27-bc92-1a05133cf5ac","year":2025},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:c547fd51ed2265dfb4b882aa7f0d1483503d0d9e7c7f7635e3e67c94235ec826","observation_id":"c9693dc2-8f6b-4e79-90e4-a082da1e3011","resolution":{"observed_at":"2026-07-10T20:37:34.711851Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.08528","last_updated":"2026-04-07T06:11:20Z","snapshot_observed_at":"2026-08-14T11:32:49.145887Z","submitted_at":"2025-04-11T13:40:53Z","title":"On The Landscape of Spoken Language Models: A Comprehensive Survey","version":2},"cited_work":{"arxiv_id":"2504.08528","doi":"10.48550/arxiv.2504.08528","metadata_source":"pith","pith_arxiv_id":"2504.08528","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"On The Landscape of Spoken Language Models: A Comprehensive Survey","venue":"cs.CL","work_id":"92746bbc-1fe6-410d-a1e9-8afab3e0a8e5","year":2025},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2504.08528","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:86f7c39842f88030d3f3658b82191cc8c4edcc45c9b45b29754e58e9f823c1e1","observation_id":"3dea55de-3b2e-4df7-8ec7-13b8c4fc7a5e","resolution":{"observed_at":"2026-07-10T20:37:34.411923Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01743","last_updated":"2025-03-07T09:05:58Z","snapshot_observed_at":"2026-08-15T22:57:45.773661Z","submitted_at":"2025-03-03T17:05:52Z","title":"Phi-4-Mini Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs","version":2},"cited_work":{"arxiv_id":"2503.01743","doi":"10.18653/v1/2023.wmt-1.23","metadata_source":"pith","pith_arxiv_id":"2503.01743","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Phi-4-Mini Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs","venue":"cs.CL","work_id":"83956045-536a-41ff-af02-b80e2a614eab","year":2025},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2503.01743","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:a79aa21b1a80e3f233af03cee842056c08a9ad38f855f682411b3cdf73f61072","observation_id":"eab642b6-1e4e-4f1b-9f97-bc41e9d4ecab","resolution":{"observed_at":"2026-07-10T20:37:34.433275Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.716881Z","title":"DeSTA: Enhancing speech language models through descriptive speech-text alignment,","venue":null,"work_id":"a9cbe219-c360-4ca9-b282-841dea23ea60","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:92c817ff69647539d5d44e2f465c4ae147ff8806be78d42bbadfb8b69714e89b","observation_id":"f01a2625-132d-4e4e-aa75-f02d060a22cf","resolution":{"observed_at":"2026-07-10T20:37:34.718414Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.704113Z","title":"Developing Instruction-Following Speech Language Model Without Speech Instruction-Tuning Data,","venue":null,"work_id":"6a33424b-8ddf-4297-b67d-edb2394304a9","year":2025},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:2b29c548cbb7c88e33bf02db3656a98c5ad28789022ee24804e787d92c66f523","observation_id":"524c9804-a579-4ad3-95df-7c8496bf23dc","resolution":{"observed_at":"2026-07-10T20:37:34.705608Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.689734Z","title":"Desta2.5- audio: Toward general-purpose large audio language model with self- generated cross-modal alignment,","venue":null,"work_id":"c50c1ee2-4fa4-4542-b522-746f4fb4d996","year":2026},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:d273d176cab1d379a8d679f42cfa486a48a86c2d80b38217baea51b612ce9f4a","observation_id":"39cc6a20-8cbc-4dab-9d92-a69221744adf","resolution":{"observed_at":"2026-07-10T20:37:34.691218Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.700238Z","title":"Prompting large language models with speech recognition abilities,","venue":null,"work_id":"99833c72-40bd-4025-adc9-08efc7c29c64","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:fdff585562dfe59bd18af47a65e97923e4ef16c357fcfd157e06baa9a523e2d3","observation_id":"6db8e2bd-cf83-472c-9e4b-ace4cdc07a2b","resolution":{"observed_at":"2026-07-10T20:37:34.701557Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.00656","last_updated":"2024-09-21T15:27:30Z","snapshot_observed_at":"2026-08-19T23:59:29.658982Z","submitted_at":"2024-03-31T12:01:32Z","title":"WavLLM: Towards Robust and Adaptive Speech Large Language Model","version":3},"cited_work":{"arxiv_id":"2404.00656","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.00656","snapshot_observed_at":"2026-07-10T20:37:34.396958Z","title":"Zhou, L","venue":"cs.CL","work_id":"440b10b8-dd06-4217-9eb2-87c1ed9e08ab","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2404.00656","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:71d0cbde9d911f2227706b11dd3e6be433bac31ef07e94eb9cccdb2c262bb9e6","observation_id":"3fe8e6f2-f109-49ef-bf16-6fcddfd29a9d","resolution":{"observed_at":"2026-07-10T20:37:34.398583Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08846","last_updated":"2024-02-13T23:25:04Z","snapshot_observed_at":"2026-08-16T14:18:48.581750Z","submitted_at":"2024-02-13T23:25:04Z","title":"An Embarrassingly Simple Approach for LLM with Strong ASR Capacity","version":1},"cited_work":{"arxiv_id":"2402.08846","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.08846","snapshot_observed_at":"2026-07-10T20:37:34.440026Z","title":"An embarrassingly simple approach for LLM with strong ASR capacity","venue":"cs.CL","work_id":"d0cf5313-bc88-4f8f-9a2c-9012a25a93dd","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2402.08846","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:dd00313540e5e8a4f70bdcfdadc8763f7427eca8bc32e8c16d201e959b52f2b8","observation_id":"3f7e0a52-5d58-43fa-926d-e60fa96128a2","resolution":{"observed_at":"2026-07-10T20:37:34.441426Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.698342Z","title":"Wav2Prompt: End-to-end speech prompt learning and task-based fine-tuning for text-based LLMs,","venue":null,"work_id":"72ebc920-7f2b-4e88-af4c-498860d470eb","year":2025},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:cab1dec277c9fc5c079f48dfbaa80cdc7482493dc2fce38e6864aa8c0378fef6","observation_id":"083a14fa-dd81-48bd-b888-c00565990f3a","resolution":{"observed_at":"2026-07-10T20:37:34.699667Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.708129Z","title":"Connecting speech encoder and large language model for asr,","venue":null,"work_id":"73ff3efb-6a1a-4daf-a9fc-4cf61b26aec0","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:a975bbb86e5595191aa53547f958050b2b1c4f1b7958ad4b679cffadb1dc18ea","observation_id":"2e081245-57f6-4186-a128-ad1f9b815477","resolution":{"observed_at":"2026-07-10T20:37:34.709409Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.05373","doi":"10.48550/arxiv.2602.05373","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Speech-xl: Towards long-form speech understanding in large speech language models,","venue":"Open MIND","work_id":"8f71cc96-bd57-46d3-ac5b-444d66001b30","year":2026},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:23f88070ceeb1dc6d4e7f38796c2e172e867609dd111421cb825d4f9374e4b69","observation_id":"3354767c-0d74-4cf4-9ad5-3cdc80b03b4f","resolution":{"observed_at":"2026-07-10T20:37:34.402206Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2604.00610","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.425227Z","title":"Speech llms are contextual reasoning transcribers,","venue":null,"work_id":"6269beb5-8ebb-4d60-8917-7a31187bfa7e","year":2026},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:15bd209760af1c1139103685699dca0411854027d4deaedc91b29d1ce5c1b838","observation_id":"93a7ca83-4585-4a38-be2e-f3f596bf8444","resolution":{"observed_at":"2026-07-10T20:37:34.427244Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.727663Z","title":"On decoder-only architecture for speech-to-text and large language model integration,","venue":null,"work_id":"b7b6be7a-c7b4-4fd9-92bc-03371cf4ec8f","year":2023},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:b34146b3a7ec4d987153ce9dc10ce97182fdc606befc11dde38d55719db921c5","observation_id":"7327d8b1-cd30-4a3f-a60d-6e21164a307c","resolution":{"observed_at":"2026-07-10T20:37:34.729136Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.20417","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.416205Z","title":"Speechmap- per: Speech-to-text embedding projector for llms,","venue":null,"work_id":"8a461ad3-6a29-4935-8f11-710db74b8d61","year":2026},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:4d797a799c3e8a663147d5e6613a8831e1ac11729b42907b5f17b657475435a9","observation_id":"6ac338bb-48c3-4918-b3fb-a060126c5936","resolution":{"observed_at":"2026-07-10T20:37:34.417901Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.751519Z","title":"Snapkv: Llm knows what you are looking for before generation,","venue":null,"work_id":"b6ef2472-6f12-4fd2-b96a-698bb466a467","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:abefa112ded74d692c10075696d29fbeb7d95cb525ae927dc6661f3828b3b091","observation_id":"b2da3027-61ab-4387-bd63-21bfc808a640","resolution":{"observed_at":"2026-07-10T20:37:34.752754Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.743507Z","title":"Minicache: Kv cache compression in depth dimension for large language models,","venue":null,"work_id":"699b9571-7a55-4c13-a979-0452e10b4c7e","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:c2914a0c1d1ebec77ed8230783c678fb0fb432123b338c87ef0ca2afa35a259e","observation_id":"56f0e0a1-1c5a-4778-8880-38c2ef289fe3","resolution":{"observed_at":"2026-07-10T20:37:34.744945Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.753444Z","title":"Open asr leaderboard: Towards reproducible and transparent multilingual and long-form speech recognition evaluation,","venue":null,"work_id":"380691eb-4acb-46cb-bd89-1afeb5539270","year":2026},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:f679bef340d7fa7b2cf2cf89856bb0d8a62376238e7d6e8bdddca4120abde689","observation_id":"705beebf-fb9f-4fa9-9e07-f513173d625a","resolution":{"observed_at":"2026-07-10T20:37:34.754734Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.741632Z","title":"Efficient memory management for large language model serving with pagedattention","venue":null,"work_id":"acd0b06c-5399-454e-bbbc-8706c0a17612","year":2023},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:bf27e6184438ac2e55cf414f72d5bd88304b577f9f29989ed843aa612c624d19","observation_id":"fb5ffe7d-8af3-4bed-96ea-283851fa926a","resolution":{"observed_at":"2026-07-10T20:37:34.742944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.692168Z","title":"Connection- ist temporal classification: labelling unsegmented sequence data with recurrent neural networks,","venue":null,"work_id":"55e8df2d-4fb4-4313-9737-5e43dcd1f8fd","year":2006},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:9552bed9d1f07614cf166ec711339a25aa312afe474e2f4dba99db883b3b5c81","observation_id":"6151166a-3fd3-4b0b-bf07-b82ebaa178a9","resolution":{"observed_at":"2026-07-10T20:37:34.693481Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.735699Z","title":"Cjst: Ctc compressor based joint speech and text training for decoder-only asr,","venue":null,"work_id":"80acd959-bcdb-4667-aabc-098354fd69ca","year":2025},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:60ea5c60f512dbec47f41e26e2159208fce3e359e9b4a5c586f9880317bc1961","observation_id":"e7e404c7-3a2f-4810-9a5c-6fe5a7be0dbe","resolution":{"observed_at":"2026-07-10T20:37:34.737191Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.17044","last_updated":"2025-06-03T10:28:07Z","snapshot_observed_at":"2026-08-16T13:15:26.613305Z","submitted_at":"2024-09-25T15:54:29Z","title":"How to Connect Speech Foundation Models and Large Language Models? What Matters and What Does Not","version":3},"cited_work":{"arxiv_id":"2409.17044","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.17044","snapshot_observed_at":"2026-07-10T20:37:34.413225Z","title":"How to Connect Speech Foundation Models and Large Language Models? What Matters and What Does Not","venue":"cs.CL","work_id":"ec5bab99-b3ae-4ce5-ba47-b532d1b929e4","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2409.17044","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:09c1d5cb4f0f9b6c6a64b5e821998eebad4caf91100cfe7b870e3f4d7ce5aab4","observation_id":"9ed1227e-f456-4693-b887-5afe4eafbb15","resolution":{"observed_at":"2026-07-10T20:37:34.414645Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.694200Z","title":"Cif: Continuous integrate-and-fire for end-to- end speech recognition,","venue":null,"work_id":"4286319e-98e0-4861-8661-442c21e6fdc9","year":2020},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:6d3489b7041eeab0c480671c791d112e2b142db87212839e1d7271529e18dc35","observation_id":"6f6605fe-fcbd-4554-836d-5b64e9564ff1","resolution":{"observed_at":"2026-07-10T20:37:34.695666Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.739823Z","title":"BLIP-2: Bootstrapping language- image pre-training with frozen image encoders and large language mod- els,","venue":null,"work_id":"c3785408-d646-4517-bc28-df1ba780d29f","year":2023},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:cd11614bfe57b8d0f42c46aa9b3b0ce15c40d2c212a0ac1645fcff244989fed5","observation_id":"aba02e2f-c556-4b4f-8ffc-e855301556bc","resolution":{"observed_at":"2026-07-10T20:37:34.741068Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02678","last_updated":"2024-10-03T17:04:48Z","snapshot_observed_at":"2026-08-21T16:42:22.094560Z","submitted_at":"2024-10-03T17:04:48Z","title":"Distilling an End-to-End Voice Assistant Without Instruction Training Data","version":1},"cited_work":{"arxiv_id":"2410.02678","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.02678","snapshot_observed_at":"2026-07-10T20:37:34.403610Z","title":"Distilling an end-to-end voice as- sistant without instruction training data","venue":"cs.CL","work_id":"9b6edce5-5624-4acc-9681-f196b3a6c0ba","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2410.02678","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:ea5909b2635271a36a01f655f9c2735de09e05a72e6a502b0ab82d46b8d766d7","observation_id":"14c54e88-8246-4c65-b344-c9b593196b10","resolution":{"observed_at":"2026-07-10T20:37:34.405190Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.747682Z","title":"A survey on large language model acceleration based on KV cache management,","venue":null,"work_id":"7ed333da-fe59-42b1-8048-76bb8e7a5661","year":2025},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:eb12ed8627915ddb091c56a5307f7a42d49d4688b79162ff75994da0ffa621ca","observation_id":"3c3ff200-163d-464e-aa09-ffc4b311661c","resolution":{"observed_at":"2026-07-10T20:37:34.748991Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.745581Z","title":"H2o: Heavy- hitter oracle for efficient generative inference of large language models,","venue":null,"work_id":"0a106c43-710e-4f9b-a06d-2ab21e657b3d","year":2023},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:30169f32f6a6087ada2ab6897299800790174ac8c50fa12895ff03b2cf04cbe4","observation_id":"c2e7dfad-44dd-419f-bb26-d97f8db5b63c","resolution":{"observed_at":"2026-07-10T20:37:34.747054Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.723518Z","title":"Efficient streaming language models with attention sinks,","venue":null,"work_id":"76698822-8675-4ff7-9106-e3d226becd41","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:d183d0dffd2160e56ebb4adc5abc6f5b884ca49163f9c4cdbd7e882bd21391d9","observation_id":"4113fdae-1bc2-41fb-aa09-0c8b85b1857e","resolution":{"observed_at":"2026-07-10T20:37:34.724774Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.725580Z","title":"Kvquant: Towards 10 million context length llm inference with kv cache quantization,","venue":null,"work_id":"f5fba8ac-3daa-4bd3-a1e1-27b7bf74df45","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:d44c9ad1765b4fa21587fbf76e42cf4090e7ed5b6e3871b1314068ca97e13dd5","observation_id":"ee2be897-38d7-4d83-bcc2-c0a7c1c5875b","resolution":{"observed_at":"2026-07-10T20:37:34.726897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.721354Z","title":"Dynamic memory compression: Retrofitting LLMs for accelerated inference,","venue":null,"work_id":"257e49e5-e2af-49b2-aba9-f7cbf756a8ae","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:23b7a5986c6b0f8aa6574cf768d0e7d8cf1eb23dbc6a448d4ffc1ad8ff67f182","observation_id":"3602818e-f029-493f-a7e6-d2610c0c9e6b","resolution":{"observed_at":"2026-07-10T20:37:34.722786Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.08454","last_updated":"2024-07-21T02:37:11Z","snapshot_observed_at":"2026-08-19T22:31:41.606680Z","submitted_at":"2024-07-11T12:50:42Z","title":"Model Tells You Where to Merge: Adaptive KV Cache Merging for LLMs on Long-Context Tasks","version":2},"cited_work":{"arxiv_id":"2407.08454","doi":"10.48550/arxiv.2407.08454","metadata_source":"pith","pith_arxiv_id":"2407.08454","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Model tells you where to merge: Adaptive kv cache merging for llms on long-context tasks","venue":"cs.CL","work_id":"37bb89e4-ffbc-414a-858a-627e6d7be94c","year":2024},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2407.08454","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:d523c4ddaf4e2d65d92ac99f6e34deebbe401d0426ce2719484841201601f004","observation_id":"0dcc80aa-866d-4b6f-8816-53c67589c206","resolution":{"observed_at":"2026-07-10T20:37:34.430154Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:15.401223+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:15.401223+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.719164Z","title":"Adapting language models to compress contexts,","venue":null,"work_id":"a88dc9e0-7234-427d-957f-d11c1e8104aa","year":2023},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:d115f07f425498315d340660a0620cc5251304ea0f5bf1d515e91995cbb85957","observation_id":"dc8867a3-fc92-451b-a903-ce4cb85dd9d9","resolution":{"observed_at":"2026-07-10T20:37:34.720595Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.03411","last_updated":"2020-12-19T09:18:21Z","snapshot_observed_at":"2026-08-15T16:45:09.638988Z","submitted_at":"2020-12-07T01:53:45Z","title":"MLS: A Large-Scale Multilingual Dataset for Speech Research","version":2},"cited_work":{"arxiv_id":"2012.03411","doi":null,"metadata_source":"pith","pith_arxiv_id":"2012.03411","snapshot_observed_at":"2026-07-10T20:37:34.434656Z","title":"MLS: A Large-Scale Multilingual Dataset for Speech Research","venue":"eess.AS","work_id":"8d788441-917e-4c1a-a15c-24d0fc471364","year":2020},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2012.03411","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:966646cbb77cd6ee040c450bdc392db3a7986758def1bef25cdb67757459ca99","observation_id":"b828db19-f4ab-4881-b9fc-778cb926758f","resolution":{"observed_at":"2026-07-10T20:37:34.436040Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.729807Z","title":"GigaSpeech: An Evolving, Multi-Domain ASR Corpus with 10,000 Hours of Transcribed Audio,","venue":null,"work_id":"7f74024b-4d49-4162-a8f0-5810fe5732e4","year":2021},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:a9e27c724fe058c5e2d718c312ef1b55521ff97423a183fb0d4b19916ba2e9f8","observation_id":"6969e927-329e-4f93-9d8b-f5e11c135e37","resolution":{"observed_at":"2026-07-10T20:37:34.731112Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.02014","last_updated":"2021-04-06T04:22:48Z","snapshot_observed_at":"2026-08-16T18:33:47.962751Z","submitted_at":"2021-04-05T17:05:28Z","title":"SPGISpeech: 5,000 hours of transcribed financial audio for fully formatted end-to-end speech recognition","version":2},"cited_work":{"arxiv_id":"2104.02014","doi":null,"metadata_source":"pith","pith_arxiv_id":"2104.02014","snapshot_observed_at":"2026-07-10T20:37:34.445713Z","title":"K., Lavrukhin, V ., Majumdar, S., Noroozi, V ., Zhang, Y ., Kuchaiev, O., Balam, J., Dovzhenko, Y ., Frey- berg, K., Shulman, M","venue":"cs.CL","work_id":"52966b84-bc53-4214-bf88-6d1462ee85de","year":2021},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2104.02014","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:bd1da2b263fb729a1cdfc57fbba68f8730ae036d9f05aaf821a95b7a344da33f","observation_id":"a262608f-fd45-4870-b55b-a47a2fde477a","resolution":{"observed_at":"2026-07-10T20:37:34.447307Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.702105Z","title":"Librispeech: an asr corpus based on public domain audio books","venue":null,"work_id":"092ed04e-8bc1-4e20-b014-e8da5ed9421e","year":2015},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:da8cdd475d669f1ad0e2360b502cbf233aa3865a5d5696b5a81d97c28df74043","observation_id":"e37a5331-43c5-4246-802e-547170de6731","resolution":{"observed_at":"2026-07-10T20:37:34.703497Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.706203Z","title":"Libriheavy: a 50,000 hours asr corpus with punctuation casing and context,","venue":null,"work_id":"9c792d98-82af-40ab-ac4b-f50cb97d92b5","year":2023},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:39b971341f2f2b9152e62da30df68d1547b0d5e9f4ad8e2e460e9798a5dd3c61","observation_id":"c15ce9cb-2b5e-4118-a982-ba21284cf367","resolution":{"observed_at":"2026-07-10T20:37:34.707462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.714704Z","title":"Common voice: A massively-multilingual speech corpus,","venue":null,"work_id":"fa5af219-10f8-4c67-a3f9-0260ea5b8311","year":2020},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:7eb3c4633428176f27a9be9ff998dfff6097773e8713e47a96414d0311994048","observation_id":"399d939d-6696-48df-8dd6-5391eb85e01d","resolution":{"observed_at":"2026-07-10T20:37:34.716163Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.733603Z","title":"V oxPopuli: A large-scale multilingual speech corpus for representation learning, semi-supervised learning and interpretation,","venue":null,"work_id":"1238442a-2394-4bbf-889a-a94e55151d9a","year":2021},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:2e109f8f6eef065a342c70e850c51de337e306872b820854cb456ca17234450d","observation_id":"c18b0848-e607-4b93-8dfb-38506e54df6a","resolution":{"observed_at":"2026-07-10T20:37:34.735106Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.696426Z","title":"Ted-lium 3: Twice as much data and corpus repartition for experiments on speaker adaptation,","venue":null,"work_id":"13cc7da3-37ff-4414-b1d5-6734687636d8","year":2018},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:b3d4e097e4dbec40f281f13a02a85dc8f9fc0cf1d465ac50495d527dcb09d207","observation_id":"ec50311e-9717-4443-b3d6-a8e24d5cce5f","resolution":{"observed_at":"2026-07-10T20:37:34.697696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.731772Z","title":"”earnings- 22: A practical benchmark for accents in the wild","venue":null,"work_id":"7b729d9b-2c01-448b-b007-24bba5aa6d7e","year":2022},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:e1bd3bc672c84329be6e011bd331195e7caa6d109adb2aac6883b84fa19c6f05","observation_id":"a3d1013e-9dd2-4e52-a9c8-ace355b79206","resolution":{"observed_at":"2026-07-10T20:37:34.733023Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.737812Z","title":"The ami meeting corpus: A pre-announcement,","venue":null,"work_id":"f5807737-594d-42a9-99fa-d7363653285d","year":2006},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:96356fb8326c7f7e512081c203a11b0373e9cc219d8973b58c4e929ae3a84087","observation_id":"8496c9cb-6b13-40a4-b382-7d1572013902","resolution":{"observed_at":"2026-07-10T20:37:34.739220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-19T10:10:54.804005Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:d54b18ef6740f73ed17564e137a93934e0c18aa4fac6f64e022168487f9a2ed9","observation_id":"fb2b9101-cb2b-4880-9cb4-6c6957330305","resolution":{"observed_at":"2026-07-10T20:37:34.444265Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.755344Z","title":"Robust speech recognition via large-scale weak super- vision,","venue":null,"work_id":"622c86e6-c64b-4089-a3fb-071ceed56df2","year":2023},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:9c9cded60c798f91ed85cbc6839b9f9000f4cb8a535205a899524e4daec2963f","observation_id":"b58aae29-800d-4348-878a-d7e8a6644d8e","resolution":{"observed_at":"2026-07-10T20:37:34.756790Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.749654Z","title":"Conformer: Convolution- augmented Transformer for Speech Recognition,","venue":null,"work_id":"b34a5725-e92b-4325-962d-6f9ed20e6ad5","year":2020},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:c7b995582581903c570da264234542cd2bd9faab0a827903952e3f09675f1bee","observation_id":"15296f6a-e947-4705-afe9-2de029452ce6","resolution":{"observed_at":"2026-07-10T20:37:34.750916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:37:34.757344Z","title":"LoRA: Low-Rank Adaptation of Large Language Models,","venue":null,"work_id":"7184de5d-3e36-46bb-8b5d-936de95b534a","year":null},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:fb61d52881a581668ec4867a3ba997f0054715f14e4531508b70672bf0ac82e1","observation_id":"ed375f2a-eaa3-45b1-a3d6-2d3d93a3b60b","resolution":{"observed_at":"2026-07-10T20:37:34.761787Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-20T18:17:38.588909Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs"},"reference_resolution":{"displayed":52,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":0,"verified_exact":16,"verified_fuzzy":34},"total_outbound_references":52},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 52 of 52 outbound references and 0 inbound Pith citation observations for arXiv:2607.06827."}