{"as_of":"2026-08-12T04:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3ff714f8de9e0db4ff15f7646012090f3ce02c724b3dec7a28a3e8bdec291bf1","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":40,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":40,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":40,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":40,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T18:08:22.102985Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-05T13:21:06.342183Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-11T18:08:22.102985Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.08237","last_updated":"2024-12-12T10:01:11Z","snapshot_observed_at":"2026-08-11T18:00:24.977636Z","submitted_at":"2024-12-11T09:38:50Z","title":"TouchTTS: An Embarrassingly Simple TTS Framework that Everyone Can Touch","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-11T18:08:22.102985Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2412.08237"},"observation_digest":"sha256:c85672223992a7537df08aaec7d23a557b518a944744a3ab127ff67ca46e8ec6","observation_id":"647a6ebd-3673-4ad8-b35c-6f63a7df5f33","resolution":{"observed_at":"2026-08-11T18:08:22.102985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-11T14:54:03.571131Z","title":"Seed- ASR : Understanding diverse speech and contexts with LLM -based speech recognition","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.11538","last_updated":"2025-09-11T05:26:13Z","snapshot_observed_at":"2026-08-11T14:47:21.523120Z","submitted_at":"2024-12-16T08:15:19Z","title":"MERaLiON-SpeechEncoder: Towards a Speech Foundation Model for Singapore and Beyond","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-11T14:54:03.571131Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2412.11538"},"observation_digest":"sha256:dcfc5dc942eb9bbfb3c87df9f0a6263c84bb8461ea798f76c07284a04cd4d525","observation_id":"cb05912a-b141-4d17-b5f9-69cd95bac048","resolution":{"observed_at":"2026-08-11T14:54:03.571131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-11T11:18:33.773017Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.15622","last_updated":"2024-12-20T07:28:04Z","snapshot_observed_at":"2026-08-12T00:25:44.448182Z","submitted_at":"2024-12-20T07:28:04Z","title":"TouchASP: Elastic Automatic Speech Perception that Everyone Can Touch","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T11:18:33.773017Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2412.15622"},"observation_digest":"sha256:62d3098b5e0cdc1d2078bc22e1277ab5edb524f137e64774f6e91fe29039e65d","observation_id":"7da03c5d-ffa7-4dad-86fa-2e57f10cecaa","resolution":{"observed_at":"2026-08-11T11:18:33.773017Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-11T10:50:48.808748Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16102","last_updated":"2025-08-09T10:01:51Z","snapshot_observed_at":"2026-08-11T10:44:57.757823Z","submitted_at":"2024-12-20T17:43:50Z","title":"Interleaved Speech-Text Language Models for Simple Streaming Text-to-Speech Synthesis","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-11T10:50:48.808748Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2412.16102"},"observation_digest":"sha256:6b1857d86923eb6ce2d52c990a8ef117f5feadf58f6a1acddfed61ec752b76ff","observation_id":"a85879f1-7564-47f2-8945-97aa42eeac38","resolution":{"observed_at":"2026-08-11T10:50:48.808748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-10T22:40:21.995444Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.01108","last_updated":"2025-01-03T08:35:34Z","snapshot_observed_at":"2026-08-10T22:32:47.754571Z","submitted_at":"2025-01-02T07:08:29Z","title":"MuQ: Self-Supervised Music Representation Learning with Mel Residual Vector Quantization","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-10T22:40:21.995444Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2501.01108"},"observation_digest":"sha256:d83771259cb23ea7827e15750193969a53838407535a1cf2faa088007c1d7818","observation_id":"b6c0bdbc-a086-47c4-b397-24b63caebdb7","resolution":{"observed_at":"2026-08-10T22:40:21.995444Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-10T14:36:19.597943Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.15177","last_updated":"2026-07-03T16:46:09Z","snapshot_observed_at":"2026-08-10T17:21:32.264725Z","submitted_at":"2025-01-25T11:15:06Z","title":"Audio-Language Models for Audio-Centric Tasks: A Systematic Survey","version":3},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-10T14:36:19.597943Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2501.15177"},"observation_digest":"sha256:b97f1959236cf5bba12fddb67c2508c70fc1c51bfc44ffa71e33eca317c7b668","observation_id":"22b1f541-060f-41ed-8f1c-ad0228b1741b","resolution":{"observed_at":"2026-08-10T14:36:19.597943Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":"2407.04675","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-05T13:21:06.342183Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":"eess.AS","work_id":"c5c60033-9068-454d-8df1-52efb011f98b","year":2024},"citing_paper":{"arxiv_id":"2503.20215","last_updated":"2025-03-26T04:17:55Z","snapshot_observed_at":"2026-08-06T08:46:20.194739Z","submitted_at":"2025-03-26T04:17:55Z","title":"Qwen2.5-Omni Technical Report","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T17:54:03.225439Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2503.20215"},"observation_digest":"sha256:d4126aeccb8adc2cc2636766037baef1bb4a1cb4f729bd0aae6d1238f31c92d7","observation_id":"bac61da4-d3b8-46ea-b07f-19f951584281","resolution":{"observed_at":"2026-05-10T17:54:03.268853Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-07T14:55:54.427250Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:54.427250Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:80feb92648c36ce83609c7f62b984282d41dc3bd3c797fdb77663e0cafa77e4a","observation_id":"888acafe-876e-438d-99b9-60c6a4d3cabd","resolution":{"observed_at":"2026-08-07T14:55:54.427250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-07T13:21:01.949770Z","title":"Seed-asr: Understanding di- verse speech and contexts with llm-based speech recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22063","last_updated":"2025-05-28T07:39:25Z","snapshot_observed_at":"2026-08-10T04:09:46.869973Z","submitted_at":"2025-05-28T07:39:25Z","title":"Weakly Supervised Data Refinement and Flexible Sequence Compression for Efficient Thai LLM-based ASR","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T13:21:01.949770Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2505.22063"},"observation_digest":"sha256:511a7c29892950bcf1b33dd5ae200b5293319d79db0a6a4ce95671e1cf275e66","observation_id":"b4e8642d-3e79-4e28-b2cc-48be220962f8","resolution":{"observed_at":"2026-08-07T13:21:01.949770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-07T13:26:50.575343Z","title":"Seed-ASR: Understanding diverse speech and contexts with LLM-based speech recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02012","last_updated":"2025-05-27T21:00:12Z","snapshot_observed_at":"2026-08-10T07:32:55.066681Z","submitted_at":"2025-05-27T21:00:12Z","title":"Leveraging Large Language Models in Visual Speech Recognition: Model Scaling, Context-Aware Decoding, and Iterative Polishing","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T13:26:50.575343Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2506.02012"},"observation_digest":"sha256:955497a6392fac5e753619bd92104d0b67294edb263d081bdab3aa93ed304c34","observation_id":"e5ce7bd1-b68d-4f4e-aaa7-1dca9bbc100b","resolution":{"observed_at":"2026-08-07T13:26:50.575343Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-07T15:27:00.351719Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11040","last_updated":"2025-05-21T04:45:11Z","snapshot_observed_at":"2026-08-07T22:34:12.973486Z","submitted_at":"2025-05-21T04:45:11Z","title":"Large Language models for Time Series Analysis: Techniques, Applications, and Challenges","version":1},"reference_index":113,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.351719Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2506.11040"},"observation_digest":"sha256:b77f65dbc2f723fd4480cf705faabd1e5c71dc47f236d3881c84fbebbb551908","observation_id":"7a4edad8-3398-4a25-ac9f-c211a6ad67ee","resolution":{"observed_at":"2026-08-07T15:27:00.351719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-07T12:09:55.809973Z","title":"Seed-ASR: Understanding diverse speech and contexts with LLM- based speech recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11064","last_updated":"2025-05-31T08:18:34Z","snapshot_observed_at":"2026-08-09T22:04:33.191967Z","submitted_at":"2025-05-31T08:18:34Z","title":"PMF-CEC: Phoneme-augmented Multimodal Fusion for Context-aware ASR Error Correction with Error-specific Selective Decoding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:09:55.809973Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2506.11064"},"observation_digest":"sha256:82fa5e2621f25ff2ebab8998e39445dbdac7c968edb13fd3dadc9a56b927aa0e","observation_id":"ba4234d8-d03c-420c-8bc8-b5ebc8443421","resolution":{"observed_at":"2026-08-07T12:09:55.809973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-07T12:09:29.130086Z","title":"Seed-ASR: Understanding di- verse speech and contexts with LLM-based speech recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12059","last_updated":"2025-05-31T07:26:44Z","snapshot_observed_at":"2026-08-10T04:52:09.969941Z","submitted_at":"2025-05-31T07:26:44Z","title":"CMT-LLM: Contextual Multi-Talker ASR Utilizing Large Language Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T12:09:29.130086Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2506.12059"},"observation_digest":"sha256:6ef2ebd10a2f927ea392e4edfbb482e35356fd12bbd877a29bf5570793bb1b99","observation_id":"a1f20979-b550-4282-b092-91f0057acbfd","resolution":{"observed_at":"2026-08-07T12:09:29.130086Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-06T16:56:51.734646Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.12252","last_updated":"2025-07-16T13:59:32Z","snapshot_observed_at":"2026-08-06T16:47:49.409409Z","submitted_at":"2025-07-16T13:59:32Z","title":"Improving Contextual ASR via Multi-grained Fusion with Large Language Models","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T16:56:51.734646Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2507.12252"},"observation_digest":"sha256:bdd6e949396fd03a20d594c0353848a296854db3972e35422f0ce82323f17e82","observation_id":"e5bf7f3a-5aa8-4e1e-9ae7-b0fc88ca08ec","resolution":{"observed_at":"2026-08-06T16:56:51.734646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":"2407.04675","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-05T13:21:06.342183Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":"eess.AS","work_id":"c5c60033-9068-454d-8df1-52efb011f98b","year":2024},"citing_paper":{"arxiv_id":"2507.16632","last_updated":"2025-08-27T16:42:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-22T14:23:55Z","title":"Step-Audio 2 Technical Report","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T05:59:50.900436Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2507.16632"},"observation_digest":"sha256:b17ec141f9eac769e98a50b44ab307f5971d68587b02635e9679b52b3b278560","observation_id":"eb5c602e-02af-4a2b-a973-a8bddc81995d","resolution":{"observed_at":"2026-05-16T05:59:51.053374Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-06T14:52:42.951815Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17527","last_updated":"2025-07-27T05:17:25Z","snapshot_observed_at":"2026-08-08T10:18:26.773169Z","submitted_at":"2025-07-23T14:07:41Z","title":"Seed LiveInterpret 2.0: End-to-end Simultaneous Speech-to-speech Translation with Your Voice","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T14:52:42.951815Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2507.17527"},"observation_digest":"sha256:274a0c303da57915605b8d6deecbbe7e2d35e1822de400d9c30854c4b6390b4d","observation_id":"dddc8553-5bbc-4af2-9566-eaf72d02e81b","resolution":{"observed_at":"2026-08-06T14:52:42.951815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-06T14:45:25.066761Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18051","last_updated":"2025-07-24T02:56:29Z","snapshot_observed_at":"2026-08-11T09:37:27.259494Z","submitted_at":"2025-07-24T02:56:29Z","title":"The TEA-ASLP System for Multilingual Conversational Speech Recognition and Speech Diarization in MLC-SLM 2025 Challenge","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:25.066761Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2507.18051"},"observation_digest":"sha256:9f8925acf80d7b914d328a5f334a661ffef6fe3ab75b69c1d4ccc7f5557bcf51","observation_id":"28c0212d-65f8-4799-9caa-4ca13afb055e","resolution":{"observed_at":"2026-08-06T14:45:25.066761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-06T14:42:56.037314Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18181","last_updated":"2025-07-28T08:55:09Z","snapshot_observed_at":"2026-08-10T18:12:48.586759Z","submitted_at":"2025-07-24T08:27:53Z","title":"SpecASR: Accelerating LLM-based Automatic Speech Recognition via Speculative Decoding","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T14:42:56.037314Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2507.18181"},"observation_digest":"sha256:8a26dddcd18b4c2c86ddd18ec98b4bbe4f720df6461f7c0bf719796a239fb2e9","observation_id":"b80dc020-9196-44e7-8b93-c1503b356253","resolution":{"observed_at":"2026-08-06T14:42:56.037314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-05T20:31:42.749315Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.10456","last_updated":"2025-08-14T08:54:01Z","snapshot_observed_at":"2026-08-10T00:37:57.356924Z","submitted_at":"2025-08-14T08:54:01Z","title":"Exploring Cross-Utterance Speech Contexts for Conformer-Transducer Speech Recognition Systems","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T20:31:42.749315Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2508.10456"},"observation_digest":"sha256:6ff1ad908ff7028d5e050030b0079e76d5ebc11c2cf4b6aca4b3ea76f4cdd26b","observation_id":"49e04d82-0591-4a2b-bcfa-165cee1248a3","resolution":{"observed_at":"2026-08-05T20:31:42.749315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-05T16:18:50.808762Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.18732","last_updated":"2025-08-26T07:00:12Z","snapshot_observed_at":"2026-08-11T14:50:43.792365Z","submitted_at":"2025-08-26T07:00:12Z","title":"Cross-Learning Fine-Tuning Strategy for Dysarthric Speech Recognition Via CDSD database","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T16:18:50.808762Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2508.18732"},"observation_digest":"sha256:1f6c4ef219c4c7fa21d508c4aee155d7f1c30a216fb7e118a9371671106ac053","observation_id":"5442749b-5a01-481a-9369-9a5ec5530831","resolution":{"observed_at":"2026-08-05T16:18:50.808762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-05T10:16:04.880310Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.04392","last_updated":"2025-09-04T17:03:58Z","snapshot_observed_at":"2026-08-09T12:04:37.130962Z","submitted_at":"2025-09-04T17:03:58Z","title":"Denoising GER: A Noise-Robust Generative Error Correction with LLM for Speech Recognition","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T10:16:04.880310Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2509.04392"},"observation_digest":"sha256:612ad93d817d94cdd5b09822e3e8edd2dba6348b70e727e02f038011bb13c8ee","observation_id":"09b12923-4b21-4f31-a03c-f2275b0f650f","resolution":{"observed_at":"2026-08-05T10:16:04.880310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-04T11:29:30.053883Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.04593","last_updated":"2026-06-08T05:49:30Z","snapshot_observed_at":"2026-08-06T08:55:28.794308Z","submitted_at":"2025-10-06T08:47:38Z","title":"UniVoice: Unifying Autoregressive ASR and Flow-Matching based TTS with Large Language Models","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-04T11:29:30.053883Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2510.04593"},"observation_digest":"sha256:efeeb0f611373f58e54df24cfb5f2c0bc08f3282860b44b6d8cb3c728205dde6","observation_id":"326f4580-d5aa-41b9-b654-8a60e83ef953","resolution":{"observed_at":"2026-08-04T11:29:30.053883Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":"2407.04675","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-05T13:21:06.342183Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":"eess.AS","work_id":"c5c60033-9068-454d-8df1-52efb011f98b","year":2024},"citing_paper":{"arxiv_id":"2603.15045","last_updated":"2026-07-07T07:07:08Z","snapshot_observed_at":"2026-08-09T22:36:02.204265Z","submitted_at":"2026-03-16T09:57:06Z","title":"LLMs and Speech: Integration vs. Combination","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-15T10:41:22.138517Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2603.15045"},"observation_digest":"sha256:e9385119e3d7818e3001300b52c8e5a242de6930b80199110dd5f627356aa21f","observation_id":"102af8f7-1d4f-4709-9e34-64fc416c88ca","resolution":{"observed_at":"2026-05-15T10:45:28.319473Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-14T20:46:17.286576Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.15045","last_updated":"2026-07-07T07:07:08Z","snapshot_observed_at":"2026-08-09T22:36:02.204265Z","submitted_at":"2026-03-16T09:57:06Z","title":"LLMs and Speech: Integration vs. Combination","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-14T20:46:17.286576Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2603.15045"},"observation_digest":"sha256:c1572905de2a273dcbc0ee0da3e62e1b558fd7eb82d05f5520a83367e25c5c6a","observation_id":"da4427f9-e0cb-4de6-9412-be9608b96132","resolution":{"observed_at":"2026-07-14T20:46:17.286576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":"2407.04675","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-05T13:21:06.342183Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":"eess.AS","work_id":"c5c60033-9068-454d-8df1-52efb011f98b","year":2024},"citing_paper":{"arxiv_id":"2604.08003","last_updated":"2026-04-09T09:07:52Z","snapshot_observed_at":"2026-08-11T03:53:44.073374Z","submitted_at":"2026-04-09T09:07:52Z","title":"Rethinking Entropy Allocation in LLM-based ASR: Understanding the Dynamics between Speech Encoders and LLMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T18:06:50.408402Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2604.08003"},"observation_digest":"sha256:26c069ea2b88811f8d4a8f9f2b099f33ab433003f285c97ad9a38e45e5f26a3b","observation_id":"3370149f-6561-4fa2-8df7-87d529d972e4","resolution":{"observed_at":"2026-05-11T05:30:57.139265Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":"2407.04675","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-05T13:21:06.342183Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":"eess.AS","work_id":"c5c60033-9068-454d-8df1-52efb011f98b","year":2024},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-11T12:40:31.299771Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:98d554342378dfce9c1c287b59e9712bdc59a2015435f720eb151cf317ce686c","observation_id":"0bae1a1b-aacb-4176-a926-4b627897d09d","resolution":{"observed_at":"2026-05-11T00:41:07.223921Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":"2407.04675","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-05T13:21:06.342183Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":"eess.AS","work_id":"c5c60033-9068-454d-8df1-52efb011f98b","year":2024},"citing_paper":{"arxiv_id":"2604.18105","last_updated":"2026-06-18T09:00:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-20T11:21:06Z","title":"NIM4-ASR: Towards Efficient, Robust, and Customizable Real-Time LLM-Based ASR","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T03:48:14.211240Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2604.18105"},"observation_digest":"sha256:085275e020761265f1872520ee4b61ea3a06c2d7f78d35a9015d32558634b2b4","observation_id":"f82fd620-648e-48fa-83ec-afae6d06c116","resolution":{"observed_at":"2026-05-11T12:21:05.820918Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":"2407.04675","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-05T13:21:06.342183Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":"eess.AS","work_id":"c5c60033-9068-454d-8df1-52efb011f98b","year":2024},"citing_paper":{"arxiv_id":"2604.18105","last_updated":"2026-06-18T09:00:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-20T11:21:06Z","title":"NIM4-ASR: Towards Efficient, Robust, and Customizable Real-Time LLM-Based ASR","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-05T13:15:50.794969Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2604.18105"},"observation_digest":"sha256:0f0d9397ab4e10e4c407937ed76910ed4d126c3aecfc0f94eb92351ab39fb280","observation_id":"7da1bb08-7c34-4060-b9c9-caab82427ab1","resolution":{"observed_at":"2026-07-05T13:21:06.346052Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":"2407.04675","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-05T13:21:06.342183Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":"eess.AS","work_id":"c5c60033-9068-454d-8df1-52efb011f98b","year":2024},"citing_paper":{"arxiv_id":"2605.02916","last_updated":"2026-04-09T03:28:46Z","snapshot_observed_at":"2026-08-03T01:34:25.525673Z","submitted_at":"2026-04-09T03:28:46Z","title":"From Synthesis to Clinical Assistance: A Strategy-Aware Agent Framework for Autism Intervention based on Real Clinical Dataset","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T17:36:58.858864Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2605.02916"},"observation_digest":"sha256:89194596a5318a2094a6fda7067c23da7af04b265952703955a6f3dabf9a5847","observation_id":"2e54763f-9c1a-474c-b355-1c5d1fa33262","resolution":{"observed_at":"2026-05-11T06:30:59.129914Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":"2407.04675","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-05T13:21:06.342183Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":"eess.AS","work_id":"c5c60033-9068-454d-8df1-52efb011f98b","year":2024},"citing_paper":{"arxiv_id":"2605.04613","last_updated":"2026-05-06T08:03:31Z","snapshot_observed_at":"2026-08-11T12:53:06.596504Z","submitted_at":"2026-05-06T08:03:31Z","title":"VocalParse: Towards Unified and Scalable Singing Voice Transcription with Large Audio Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-08T16:59:25.973854Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2605.04613"},"observation_digest":"sha256:912fb693445068383cc853f8d8a90e48dd9133f6602b7929c87acd73646229c4","observation_id":"bcd6b14d-8c89-4547-97c8-52ce9ab99065","resolution":{"observed_at":"2026-05-11T17:51:09.717541Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":"2407.04675","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-05T13:21:06.342183Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":"eess.AS","work_id":"c5c60033-9068-454d-8df1-52efb011f98b","year":2024},"citing_paper":{"arxiv_id":"2605.16896","last_updated":"2026-05-16T09:16:09Z","snapshot_observed_at":"2026-08-07T13:59:36.831621Z","submitted_at":"2026-05-16T09:16:09Z","title":"JSPG: Dynamic Dictionary Filtering via Joint Semantic-Pinyin-Glyph Retrieval for Chinese Contextual ASR","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-19T20:56:04.778077Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2605.16896"},"observation_digest":"sha256:62a30105fc13551ae4fdfd4f1918be0f3e46a8ffdf3c5e9c8acf64213f31eb39","observation_id":"d9ededf9-0e59-4227-8d4e-c17da6fba2ce","resolution":{"observed_at":"2026-05-19T20:57:46.907545Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":"2407.04675","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-05T13:21:06.342183Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":"eess.AS","work_id":"c5c60033-9068-454d-8df1-52efb011f98b","year":2024},"citing_paper":{"arxiv_id":"2605.23463","last_updated":"2026-05-22T10:24:50Z","snapshot_observed_at":"2026-08-02T21:50:12.344912Z","submitted_at":"2026-05-22T10:24:50Z","title":"StepAudio 2.5 Technical Report","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-25T02:52:22.610397Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2605.23463"},"observation_digest":"sha256:0fc98a0176d4fb391715dd77d0410383659702b1289529e753d2994793b7fe33","observation_id":"91001d88-a7f2-4930-8541-dc76486a95f5","resolution":{"observed_at":"2026-05-25T02:55:16.515698Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":"2407.04675","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-05T13:21:06.342183Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":"eess.AS","work_id":"c5c60033-9068-454d-8df1-52efb011f98b","year":2024},"citing_paper":{"arxiv_id":"2606.05121","last_updated":"2026-06-03T17:26:11Z","snapshot_observed_at":"2026-08-02T17:56:34.317060Z","submitted_at":"2026-06-03T17:26:11Z","title":"Audio Interaction Model","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-28T04:57:05.062465Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2606.05121"},"observation_digest":"sha256:14f0f78ddee84b16f9d7963dc17f7ff53919aad0c6de0ce898be6364783b0ed3","observation_id":"42e76763-52ed-49d5-94cd-e443283301dd","resolution":{"observed_at":"2026-07-02T10:46:52.389064Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":"2407.04675","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-05T13:21:06.342183Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":"eess.AS","work_id":"c5c60033-9068-454d-8df1-52efb011f98b","year":2024},"citing_paper":{"arxiv_id":"2606.08486","last_updated":"2026-06-07T07:15:34Z","snapshot_observed_at":"2026-07-06T23:47:57.376596Z","submitted_at":"2026-06-07T07:15:34Z","title":"TRADE: Transducer-Augmented Decoder for Speech LLM","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-06-27T18:40:19.688550Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2606.08486"},"observation_digest":"sha256:b504054b0da4424416673a5421e0002d113cf38cefeb0b514fb497ee46b6ddd5","observation_id":"1044f1c1-3172-47e7-a2df-c706a6340555","resolution":{"observed_at":"2026-07-02T22:47:25.686348Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":"2407.04675","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-05T13:21:06.342183Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":"eess.AS","work_id":"c5c60033-9068-454d-8df1-52efb011f98b","year":2024},"citing_paper":{"arxiv_id":"2607.01733","last_updated":"2026-07-02T05:42:01Z","snapshot_observed_at":"2026-08-06T21:21:57.811880Z","submitted_at":"2026-07-02T05:42:01Z","title":"Rethinking Speech-LLM Integration for ASR: Effective Joint Speech-Text Training by Interleaving","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-03T15:17:52.966144Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2607.01733"},"observation_digest":"sha256:abef2d8c5ab61e78ba6093ed74052bb18382c6197f1f12326bb6a0d8b3855e62","observation_id":"e18ee8f2-5b9b-449a-9983-987079b5e7ee","resolution":{"observed_at":"2026-07-03T15:18:32.637916Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-11T09:26:54.286790Z","title":"Seed- ASR: Understanding diverse speech and contexts with LLM- based speech recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.05058","last_updated":"2026-07-06T13:31:09Z","snapshot_observed_at":"2026-08-03T15:27:29.701755Z","submitted_at":"2026-07-06T13:31:09Z","title":"Context-Aware ASR for Mandarin Technical Lectures","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-11T09:26:54.286790Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2607.05058"},"observation_digest":"sha256:38f4fec5da96cef2ffcf613e17dc63d786aab8679431bbfcb5fb183f044dcbe7","observation_id":"4641fd45-9eeb-4ad6-84e6-8c6a34f9414e","resolution":{"observed_at":"2026-07-11T09:26:54.286790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-03T10:13:54.253318Z","title":"arXiv preprint arXiv:2407.04675 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29279","last_updated":"2026-07-31T10:50:14Z","snapshot_observed_at":"2026-08-06T04:02:53.771607Z","submitted_at":"2026-07-31T10:50:14Z","title":"ParaASR: Multi-Token Prediction for Fast and Long-Context LLM-Based Speech Recognition","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-03T10:13:54.253318Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2607.29279"},"observation_digest":"sha256:388afeb1bf8f15a2c280976e52a95a905ce3bd00162423b7646bf5c0e2965125","observation_id":"7d9304c1-3dee-415d-92b3-b42649a97cf5","resolution":{"observed_at":"2026-08-03T10:13:54.253318Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-06T00:31:25.062446Z","title":"Seed-ASR: Understanding diverse speech and contexts with LLM-based speech recognition.arXiv preprint arXiv:2407.04675, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.01157","last_updated":"2026-08-02T11:28:13Z","snapshot_observed_at":"2026-08-08T03:59:35.155645Z","submitted_at":"2026-08-02T11:28:13Z","title":"InteracVid: Building a Real Interactive Audio-Visual Response Dataset from Live-Chat Videos","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T00:31:25.062446Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2608.01157"},"observation_digest":"sha256:71ac45e3094a53a194bf9c5bbc31c8d7585c30b3f3905f4bda7406ef41e52ba0","observation_id":"85433cbc-b637-4948-8f02-ace2f144e042","resolution":{"observed_at":"2026-08-06T00:31:25.062446Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-05T16:07:18.513493Z","title":"Seed-asr: Understanding di- verse speech and contexts with llm-based speech recog- nition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.03610","last_updated":"2026-08-10T11:37:39Z","snapshot_observed_at":"2026-08-12T04:17:33.629392Z","submitted_at":"2026-08-04T13:02:47Z","title":"Language-Specialized Multi-Teacher On-Policy Distillation for Multilingual LLM-Based ASR","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T16:07:18.513493Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2608.03610"},"observation_digest":"sha256:4e2b73d50b587abdd50875184f41b615e978f793c47d4f7efdb02de9ee546ca5","observation_id":"071cbf86-bfd9-4d4e-9c88-3ad455cd2ed5","resolution":{"observed_at":"2026-08-05T16:07:18.513493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-11T04:22:00.929409Z","title":"Seed-ASR: Understanding diverse speech and contexts with llm-based speech recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.03610","last_updated":"2026-08-10T11:37:39Z","snapshot_observed_at":"2026-08-12T04:17:33.629392Z","submitted_at":"2026-08-04T13:02:47Z","title":"Language-Specialized Multi-Teacher On-Policy Distillation for Multilingual LLM-Based ASR","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T04:22:00.929409Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2608.03610"},"observation_digest":"sha256:400634738cbd7b0595ba5f67b4556b2903f5cb5e383af57d9d3e565d9c82e423","observation_id":"d5cd0330-cb34-4039-8265-c9ab44e1570d","resolution":{"observed_at":"2026-08-11T04:22:00.929409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2407.04675/citation-record","integrity":"/paper/2407.04675/integrity","json":"/paper/2407.04675/citation-record.json","paper":"/paper/2407.04675"},"outbound":[],"paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","latest_version":2,"primary_category":"eess.AS","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 40 inbound Pith citation observations for arXiv:2407.04675."}