{"as_of":"2026-08-08T02:29:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8e84c491b49d47c4bf91198e99838abce8e2cac4676fdd72a736cba35e8fa170","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:26:23.320594Z","state":"measured"},{"denominator":41,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":41,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:26:22.389818Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-21T07:39:48.680240Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.02457","snapshot_observed_at":"2026-08-07T11:26:22.389818Z","title":"Recent advancements in large language models (LLMs) have led to remarkable breakthroughs in speech LLMs and voice assistants","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.389818Z"},"links":{"cited_paper":"/paper/2506.02457","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:8946c1327029bf96aedf0f313adfc5a8f4b51f147b8cd16f13d283be13741748","observation_id":"24a79a7e-e7f3-46b6-8e29-f2eabedc2326","resolution":{"observed_at":"2026-08-07T11:26:22.389818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"cited_work":{"arxiv_id":"2506.02457","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.02457","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sova-bench: Benchmarking the speech con- versation ability for llm-based voice assistant","venue":null,"work_id":"46008301-0f06-45f2-952f-b38d21b03566","year":2025},"citing_paper":{"arxiv_id":"2605.20266","last_updated":"2026-05-18T20:21:32Z","snapshot_observed_at":"2026-08-03T05:12:45.221230Z","submitted_at":"2026-05-18T20:21:32Z","title":"A Survey of Large Audio Language Models: Generalization, Trustworthiness, and Outlook","version":1},"reference_index":190,"source":"pdf_text","source_observed_at":"2026-05-21T07:38:23.099479Z"},"links":{"cited_paper":"/paper/2506.02457","citing_paper":"/paper/2605.20266"},"observation_digest":"sha256:5f787aaedaec6ab84e4aeebc878dd1af224aa3b99fce086df2accb627f098b58","observation_id":"77213481-1a76-4b50-ab3d-5c2d1777249c","resolution":{"observed_at":"2026-05-21T07:39:48.681853Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.02457/citation-record","integrity":"/paper/2506.02457/integrity","json":"/paper/2506.02457/citation-record.json","paper":"/paper/2506.02457"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.02457","snapshot_observed_at":"2026-08-07T11:26:22.389818Z","title":"Recent advancements in large language models (LLMs) have led to remarkable breakthroughs in speech LLMs and voice assistants","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.389818Z"},"links":{"cited_paper":"/paper/2506.02457","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:8946c1327029bf96aedf0f313adfc5a8f4b51f147b8cd16f13d283be13741748","observation_id":"24a79a7e-e7f3-46b6-8e29-f2eabedc2326","resolution":{"observed_at":"2026-08-07T11:26:22.389818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:26.672051Z","title":"Speech LLM Speech LLM extends the understanding capability to speech flow, performing modality alignment between speech and text via an encoder with adaptors","venue":null,"work_id":"be030667-7682-4a52-ab25-3a54357b5e8e","year":null},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.451786Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:2240bc309856ab5284a9f3c42dfdefc2ca6cff65d2210667c938b2f59d8472c1","observation_id":"6935127e-fa43-4f0d-b77e-26f4d1a3a331","resolution":{"observed_at":"2026-08-07T11:26:26.828263Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:26.554859Z","title":null,"venue":null,"work_id":"40d028fb-0da7-4556-8190-d6fa0a17fd76","year":null},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.535835Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:ab0a9f94648a62c5ca57820b4d72601357ef25b7e577f0e0a8724db9e6e198cf","observation_id":"335ffbc3-ee3e-4578-a789-24643a2af936","resolution":{"observed_at":"2026-08-07T11:26:26.608938Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"1637.8469","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:23.775478Z","title":null,"venue":null,"work_id":"7539834e-9728-44f1-9575-428c65a03879","year":null},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.625987Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:ac3a5e956c65bf5417e60e07133a493e4bffecb4a3126a10720709b051a39311","observation_id":"3007e468-ce9c-436f-8f31-d8c14bf5e103","resolution":{"observed_at":"2026-08-07T11:26:23.863004Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:26.395268Z","title":null,"venue":null,"work_id":"0b34e5c9-e528-4844-a1fc-b68fbb04ecff","year":null},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.693809Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:b4caaba8014be83f1edca0220ea351dd544a132db06c8e9686aa6c4c5581bb82","observation_id":"741100b8-5df6-4d9e-8de8-ec39610b01af","resolution":{"observed_at":"2026-08-07T11:26:26.436653Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:26.230509Z","title":null,"venue":null,"work_id":"216a0ee5-d1e3-476a-a20d-9701448a08a3","year":null},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.770624Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:f81892001837b631e7d666a45d6ac9ae710c0cb37f3d75fc8ec9d6f3da23bdcd","observation_id":"eed943fb-c74a-42e8-86bb-cac7be6255bb","resolution":{"observed_at":"2026-08-07T11:26:26.295669Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-07T11:26:22.870915Z","title":"GPT-4o system card,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.870915Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:87e63f5e9b7a97aacaa37f585f9d7399bc229aabb22b52fba5266e9e0487c195","observation_id":"04f99b43-0c47-40cd-b373-185506acfab5","resolution":{"observed_at":"2026-08-07T11:26:22.870915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.16725","last_updated":"2024-11-05T02:24:18Z","snapshot_observed_at":"2026-07-06T19:07:46.545514Z","submitted_at":"2024-08-29T17:18:53Z","title":"Mini-Omni: Language Models Can Hear, Talk While Thinking in Streaming","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.16725","snapshot_observed_at":"2026-08-07T11:26:22.926836Z","title":"Mini-Omni: Language models can hear, talk while thinking in streaming,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.926836Z"},"links":{"cited_paper":"/paper/2408.16725","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:4eb923c1b24266e6b31aee39cd5e2e2dad95566bdf7f6d5320a7b9636fed8b5d","observation_id":"b1ef39e5-7134-495b-aee8-c300409e6154","resolution":{"observed_at":"2026-08-07T11:26:22.926836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.06666","last_updated":"2025-03-01T12:59:49Z","snapshot_observed_at":"2026-07-06T19:13:20.458958Z","submitted_at":"2024-09-10T17:34:34Z","title":"LLaMA-Omni: Seamless Speech Interaction with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.06666","snapshot_observed_at":"2026-08-07T11:26:23.008363Z","title":"LLaMA-Omni: Seamless speech interaction with large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.008363Z"},"links":{"cited_paper":"/paper/2409.06666","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:0d6ac555e47ea736f8e3834d9a6fe38c859df66d3bc7ec27ad04ccf0137b555f","observation_id":"3a38c437-6768-4ac8-b6b1-defd92aa8332","resolution":{"observed_at":"2026-08-07T11:26:23.008363Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00037","last_updated":"2024-10-02T09:11:45Z","snapshot_observed_at":"2026-07-30T10:21:14.474746Z","submitted_at":"2024-09-17T17:55:39Z","title":"Moshi: a speech-text foundation model for real-time dialogue","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00037","snapshot_observed_at":"2026-08-07T11:26:23.050878Z","title":"Moshi: a speech-text foundation model for real-time dialogue,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.050878Z"},"links":{"cited_paper":"/paper/2410.00037","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:47294e4127033844288701bbc4e89dfd2299a5027e98c541d6c476eb462fcff3","observation_id":"f821b026-4326-4f47-b148-bb854874ed7f","resolution":{"observed_at":"2026-08-07T11:26:23.050878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:26.110174Z","title":"Dynamic- SUPERB: Towards a dynamic, collaborative, and comprehensive instruction-tuning benchmark for speech,","venue":null,"work_id":"04713159-10a5-4961-93ac-51dfc03fff55","year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.138843Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:1dbf68a5365b96e7c25dae3cb96cacd9576edea91a73011f79e0a3bdf5f5bdd2","observation_id":"657ac16c-0126-4684-af44-6e56729f7170","resolution":{"observed_at":"2026-08-07T11:26:26.163525Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.05361","last_updated":"2025-06-09T16:36:12Z","snapshot_observed_at":"2026-07-06T19:47:16.854192Z","submitted_at":"2024-11-08T06:33:22Z","title":"Dynamic-SUPERB Phase-2: A Collaboratively Expanding Benchmark for Measuring the Capabilities of Spoken Language Models with 180 Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.05361","snapshot_observed_at":"2026-08-07T11:26:23.197926Z","title":"Dynamic- SUPERB Phase-2: A collaboratively expanding benchmark for measuring the capabilities of spoken language models with 180 tasks,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.197926Z"},"links":{"cited_paper":"/paper/2411.05361","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:7d63e4497098ddfed1b4312932c8dd15ef07c271dc7eb397a1bdfec161bb0074","observation_id":"5aad3ced-54b1-4a76-8309-61beb49207e2","resolution":{"observed_at":"2026-08-07T11:26:23.197926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16020","last_updated":"2025-05-06T00:52:19Z","snapshot_observed_at":"2026-08-07T08:31:36.184375Z","submitted_at":"2024-06-23T05:40:26Z","title":"AudioBench: A Universal Benchmark for Audio Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16020","snapshot_observed_at":"2026-08-07T11:26:23.203188Z","title":"AudioBench: A universal benchmark for audio large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.203188Z"},"links":{"cited_paper":"/paper/2406.16020","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:f871f096302410d039d81ff97fd42e1dd2b6e1145bff7bacda4a9266cae0f7cf","observation_id":"0db694d5-841a-41cf-99e1-1c9efe1a8c85","resolution":{"observed_at":"2026-08-07T11:26:23.203188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07729","last_updated":"2024-07-26T06:30:47Z","snapshot_observed_at":"2026-08-06T03:59:19.956442Z","submitted_at":"2024-02-12T15:41:22Z","title":"AIR-Bench: Benchmarking Large Audio-Language Models via Generative Comprehension","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07729","snapshot_observed_at":"2026-08-07T11:26:23.208362Z","title":"AIR-Bench: Benchmarking large audio- language models via generative comprehension,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.208362Z"},"links":{"cited_paper":"/paper/2402.07729","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:45ab51131a01d869c98996daf408db47b99f56a1849019367873833edbb8100c","observation_id":"1dae2d60-7300-4edf-ad27-3acb7069dff2","resolution":{"observed_at":"2026-08-07T11:26:23.208362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17196","last_updated":"2024-12-11T15:45:21Z","snapshot_observed_at":"2026-07-06T19:37:56.214143Z","submitted_at":"2024-10-22T17:15:20Z","title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17196","snapshot_observed_at":"2026-08-07T11:26:23.213004Z","title":"V oiceBench: Benchmarking llm-based voice assistants,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.213004Z"},"links":{"cited_paper":"/paper/2410.17196","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:9bde6de724e2523efcc1390e5c3d0b3a6661873cd5a6407466f7cf971bc1a4c1","observation_id":"277382e1-99a2-4fc9-ab1b-e2432272bb66","resolution":{"observed_at":"2026-08-07T11:26:23.213004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:25.988780Z","title":"SALMONN:Towards generic hearing abilities for large language models,","venue":null,"work_id":"a64edb40-dbba-48f3-a0cc-543679775d0e","year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.217381Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:f46f65eed3ea63d6fd720b301d30d2429dc8fd4b9665cef8dcb18f144dd5c264","observation_id":"3208b10f-2a13-4b05-85bf-ef46656c5d91","resolution":{"observed_at":"2026-08-07T11:26:26.018205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:25.816990Z","title":"SpeechGPT: Empowering large language models with intrinsic cross-modal conversational abilities,","venue":null,"work_id":"efb3446f-1bfc-4556-bbdd-824c66ee2d2b","year":2023},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.222170Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:aef345882f51a396e54fb0ac881c2724404d04dcb747ea158b6343aba8d3bee7","observation_id":"c243da26-5a10-41da-9c47-b0fc12b019fb","resolution":{"observed_at":"2026-08-07T11:26:25.916923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10759","last_updated":"2024-07-15T14:38:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T14:38:09Z","title":"Qwen2-Audio Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10759","snapshot_observed_at":"2026-08-07T11:26:23.226286Z","title":"Qwen2-audio technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.226286Z"},"links":{"cited_paper":"/paper/2407.10759","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:52d6b321a255714b0ec062f8f96d4bbf391115a03d4d5033bb329c4a6c0db443","observation_id":"2f99f1ab-e977-4b39-b9f2-dc13e6840bf7","resolution":{"observed_at":"2026-08-07T11:26:23.226286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:25.701614Z","title":"SNAC: Multi- scale neural audio codec,","venue":null,"work_id":"28b5eb90-4bb6-4f96-b4fb-bd1ef92efff0","year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.230921Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:e0d84845eda3ead41fce258d3d33a97d14d2fc64184c3ce45abca83e450effbd","observation_id":"d72b60f0-b1c0-4ad4-86ed-f944745a5b04","resolution":{"observed_at":"2026-08-07T11:26:25.756587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:23.234959Z","title":"Hubert: Self-supervised speech represen- tation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.234959Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:548d47b6f01305ced45ef56abdcb9b7cdd393e508563ac20970c9fa7fd7ada3f","observation_id":"78bf5a99-e0a4-4aac-ba3e-64c54fd2a8ab","resolution":{"observed_at":"2026-08-07T11:26:23.234959Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:25.559866Z","title":"HiFi-GAN: Generative adversar- ial networks for efficient and high fidelity speech synthesis,","venue":null,"work_id":"80bfa388-5d05-4e75-ad84-03d9e4a106a4","year":2020},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.239575Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:690cad50bd543718c00da9c29c746162335faeeafe97a0e3464d8839f16b3014","observation_id":"25d6d93a-a35b-43e3-a76c-698b211e4008","resolution":{"observed_at":"2026-08-07T11:26:25.610697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:25.369020Z","title":"Speech resynthesis from discrete disentangled self-supervised representations,","venue":null,"work_id":"65a43c59-4527-46af-a95d-69d954c59a65","year":2021},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.244104Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:8117e9b6cba774ebd0d07a8dbe6604c1471b400909671ae4accebbd507980bea","observation_id":"81babd28-831b-43ef-be8d-cef09381c961","resolution":{"observed_at":"2026-08-07T11:26:25.430775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:25.164843Z","title":"Westlake-Omni,","venue":null,"work_id":"0b3bec86-a6d0-4bc3-be74-df9898f0f636","year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.248523Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:52506912405766e1163f868ffb71dbe06c3a3c5c4103d9f681a1befaf3aab964","observation_id":"37d13974-9f50-4609-bf99-3a907d1da2c4","resolution":{"observed_at":"2026-08-07T11:26:25.265736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00774","last_updated":"2024-12-08T05:41:56Z","snapshot_observed_at":"2026-08-04T02:03:05.931925Z","submitted_at":"2024-11-01T17:59:51Z","title":"Freeze-Omni: A Smart and Low Latency Speech-to-speech Dialogue Model with Frozen LLM","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.00774","snapshot_observed_at":"2026-08-07T11:26:23.253202Z","title":"Freeze- Omni: A smart and low latency speech-to-speech dialogue model with frozen llm,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.253202Z"},"links":{"cited_paper":"/paper/2411.00774","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:19a384b8a07dd8bd3a09d3c541012698abdb5e35cbab688b8c4e3b4efd8a15b4","observation_id":"fd4d79c4-b7f4-466a-bb6d-a80d9ec59e96","resolution":{"observed_at":"2026-08-07T11:26:23.253202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:24.969618Z","title":"Be- yond turn-based interfaces: Synchronous llms as full-duplex dia- logue agents,","venue":null,"work_id":"f2f46819-3e67-4bcf-ad9a-67f5aaa1d25f","year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.257541Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:c1d40968b4ee46d0b5a5c0dd6bd7f1e46e1bdcffd8d5ec6602a29d1453a4ddca","observation_id":"f725328a-7a76-49dc-a579-cb24b7ecc36c","resolution":{"observed_at":"2026-08-07T11:26:25.096745Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17799","last_updated":"2025-01-03T06:15:58Z","snapshot_observed_at":"2026-07-06T19:38:26.783620Z","submitted_at":"2024-10-23T11:58:58Z","title":"OmniFlatten: An End-to-end GPT Model for Seamless Voice Conversation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17799","snapshot_observed_at":"2026-08-07T11:26:23.261871Z","title":"OmniFlatten: An end-to-end GPT model for seamless voice conversation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.261871Z"},"links":{"cited_paper":"/paper/2410.17799","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:51a4df470936f024f0dae117865e9b297ccba25ad051007c905555cd90be1c30","observation_id":"101c2dd5-5392-419c-820a-30b3c3d19b08","resolution":{"observed_at":"2026-08-07T11:26:23.261871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:23.266830Z","title":"Baichuan-Omni-1.5 technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.266830Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:5a3f0b19e4fd6e1df81186dc37f2c41422fd24bff8a19edf54ab3892d7d8357c","observation_id":"b577a389-a753-4102-adbf-4c98624ca753","resolution":{"observed_at":"2026-08-07T11:26:23.266830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:24.807192Z","title":"TriviaQA: A large scale distantly supervised challenge dataset for reading comprehension,","venue":null,"work_id":"3a347501-a554-4349-92be-e1ae803d62df","year":2017},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.271445Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:35b0821e2a2cb275f9a7c214bc396cfc8163ded26f806db8eafa7bf5a2143b9b","observation_id":"f9735e47-eb7d-479c-a8ac-03dc3d0f8cf8","resolution":{"observed_at":"2026-08-07T11:26:24.898400Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:23.276005Z","title":"Lib- rispeech: an asr corpus based on public domain audio books,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.276005Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:36ef61a78d373420d66731934a3217a084018736f7a4a6f953c41c6c409841d4","observation_id":"2733f0ee-bbe7-44dd-b717-497f1ce9d094","resolution":{"observed_at":"2026-08-07T11:26:23.276005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:24.633120Z","title":"LibriSQA: A novel dataset and framework for spoken question answering with large language models,","venue":null,"work_id":"02957620-88da-4e6a-9150-ab0d27bd6dbd","year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.280776Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:f0498bb37726212ed657b1b0a0bc4fff21036e016aa430828a2cae6efb3c9f0f","observation_id":"ffa586ec-7787-4748-bc32-a3b4aa3794d3","resolution":{"observed_at":"2026-08-07T11:26:24.706381Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:24.498269Z","title":"Spoken SQuAD: A study of mitigating the impact of speech recognition errors on listening comprehension,","venue":null,"work_id":"3f2b54c7-a558-42f5-ae18-026cd5094027","year":2018},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.285411Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:1966e23182059dbb8c5c1d55259cd0a3ab14e9683e394a2a324b550f506d70c5","observation_id":"4ad395e0-aac3-43e9-820e-feafba8e888d","resolution":{"observed_at":"2026-08-07T11:26:24.560476Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:23.289433Z","title":"IEMOCAP: Interactive emotional dyadic motion capture database,","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.289433Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:8c29289263313dc3592db5d003d2e861ff7f87ee08efd59063ee41871b1f5261","observation_id":"6d061c8f-99cc-4c34-97b1-977e6766fafb","resolution":{"observed_at":"2026-08-07T11:26:23.289433Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:24.237193Z","title":"Common V oice: A massively-multilingual speech corpus,","venue":null,"work_id":"cea79bc6-bd9e-4ed8-8ae0-6983fe939044","year":2020},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.294124Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:3542c9ec60f4a92422ec0682db1a0a40a530b67fd40fc4f114927be23e7f9828","observation_id":"3f915456-202c-420d-b2bd-e2b094f68824","resolution":{"observed_at":"2026-08-07T11:26:24.374900Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:24.035097Z","title":"Stanford alpaca: an instruction- following llama model (2023),","venue":null,"work_id":"1601a21f-7e25-4ec2-bbd7-bea8804dd428","year":2023},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.298170Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:edf3a45376fc6cc2fec9f90a6cf53d612c398ac3a6ffad01b60028b0697fc5b9","observation_id":"16d4b01d-6eed-45ce-aa65-308b97d668b2","resolution":{"observed_at":"2026-08-07T11:26:24.140936Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.09305","last_updated":"2024-09-14T05:03:18Z","snapshot_observed_at":"2026-07-06T19:15:17.200627Z","submitted_at":"2024-09-14T05:03:18Z","title":"The T05 System for The VoiceMOS Challenge 2024: Transfer Learning from Deep Image Classifier to Naturalness MOS Prediction of High-Quality Synthetic Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.09305","snapshot_observed_at":"2026-08-07T11:26:23.302596Z","title":"The t05 sys- tem for the voicemos challenge 2024: Transfer learning from deep image classifier to naturalness mos prediction of high-quality syn- thetic speech,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.302596Z"},"links":{"cited_paper":"/paper/2409.09305","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:7b8e9293c57fd4e7d05a1f620a5ade765f9b49022b2daba8ffee705525b6e8b2","observation_id":"3ca2bfd7-9187-4d2a-945d-c5e9210fa9d0","resolution":{"observed_at":"2026-08-07T11:26:23.302596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-07-06T18:42:34.958119Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-07T11:26:23.307131Z","title":"Cosyvoice: A scalable multi- lingual zero-shot text-to-speech synthesizer based on supervised semantic tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.307131Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:c597fb62c29eb848bf164930d4cc332794fbd7b3f92fb90d34fdaf31d9d3c09c","observation_id":"f82a96c8-3bec-43ff-899d-1ad5b1167973","resolution":{"observed_at":"2026-08-07T11:26:23.307131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:23.311743Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.311743Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:0d55415c91a2a343660a565400a329229e7dcd266bb1019aef70ab4fe6fc9179","observation_id":"3141e860-23e4-4cdc-b2df-98ae5a89919b","resolution":{"observed_at":"2026-08-07T11:26:23.311743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11190","last_updated":"2024-11-05T02:27:57Z","snapshot_observed_at":"2026-07-06T19:33:34.819837Z","submitted_at":"2024-10-15T02:10:45Z","title":"Mini-Omni2: Towards Open-source GPT-4o with Vision, Speech and Duplex Capabilities","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11190","snapshot_observed_at":"2026-08-07T11:26:23.315974Z","title":"Mini-Omni2: Towards open-source GPT- 4o with vision, speech and duplex capabilities,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.315974Z"},"links":{"cited_paper":"/paper/2410.11190","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:d4f87e0acaff8d260304f42658fa38973269e2cd49be6fd9c738ee85db45ab2f","observation_id":"4b7c8061-2bde-426b-8c21-ff56e53f4e71","resolution":{"observed_at":"2026-08-07T11:26:23.315974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02612","last_updated":"2024-12-03T17:41:24Z","snapshot_observed_at":"2026-08-02T21:27:26.884251Z","submitted_at":"2024-12-03T17:41:24Z","title":"GLM-4-Voice: Towards Intelligent and Human-Like End-to-End Spoken Chatbot","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02612","snapshot_observed_at":"2026-08-07T11:26:23.320594Z","title":"GLM-4-V oice: Towards intelligent and human-like end- to-end spoken chatbot,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.320594Z"},"links":{"cited_paper":"/paper/2412.02612","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:cc7fad1fee9039d2dc0388d39a77100b9607edef7e2b81f32aa91049da3933f5","observation_id":"8b15efbd-c7c1-496d-b92e-9470a0a2afa2","resolution":{"observed_at":"2026-08-07T11:26:23.320594Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-07T11:21:04.479511Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":24,"verified_exact":0,"verified_fuzzy":14},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 2 inbound Pith citation observations for arXiv:2506.02457."}