{"as_of":"2026-08-07T18:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:59dc808c5c87d2b3fc41d7aae71c525cca3509d3008b367d6973908b420c2dc5","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":17,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T10:36:06.472390Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2505.15957","last_updated":"2026-04-26T17:05:53Z","snapshot_observed_at":"2026-07-06T21:28:06.120576Z","submitted_at":"2025-05-21T19:17:29Z","title":"Towards Holistic Evaluation of Large Audio-Language Models: A Comprehensive Survey","version":4},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-22T13:32:57.771753Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2505.15957"},"observation_digest":"sha256:ff15eaa377d0bc23a2407cd6d7917e7c673466b32a23930d702364bfa18c74ff","observation_id":"db7cb67f-4973-4c2f-9e03-74dc6668fbff","resolution":{"observed_at":"2026-05-22T13:34:53.242696Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2507.16632","last_updated":"2025-08-27T16:42:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-22T14:23:55Z","title":"Step-Audio 2 Technical Report","version":3},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-16T05:59:50.900436Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2507.16632"},"observation_digest":"sha256:a27f55d882a2a3b34e24c218800141eff9ee925655bd177baa27fb13e49cb0b4","observation_id":"106fd36e-5574-4659-a943-269086e2d72c","resolution":{"observed_at":"2026-05-16T05:59:51.100939Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T10:36:06.472390Z","title":"Zeng, A.; Du, Z.; Liu, M.; Wang, K.; Jiang, S.; Zhao, L.; Dong, Y .; and Tang, J","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.03940","last_updated":"2025-09-04T07:03:46Z","snapshot_observed_at":"2026-08-06T18:04:23.478268Z","submitted_at":"2025-09-04T07:03:46Z","title":"VoxRole: A Comprehensive Benchmark for Evaluating Speech-Based Role-Playing Agents","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T10:36:06.472390Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2509.03940"},"observation_digest":"sha256:6985e81c7464e225a871795c25e4494738a3b810beaf647e346f2e24cd25417a","observation_id":"d17a3726-3a7b-4486-a695-6b0d4d77566b","resolution":{"observed_at":"2026-08-05T10:36:06.472390Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2509.26388","last_updated":"2026-05-01T17:28:37Z","snapshot_observed_at":"2026-07-06T22:31:15.349221Z","submitted_at":"2025-09-30T15:23:39Z","title":"Game-Time: Evaluating Temporal Dynamics in Spoken Language Models","version":4},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-18T11:51:43.561210Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2509.26388"},"observation_digest":"sha256:47507c8283607d39c8e7de1923cc728194975b9b9a0c0d1f2410e10a097c3e9d","observation_id":"ccd32e67-ce37-47b6-92ba-2e7f39352a61","resolution":{"observed_at":"2026-05-18T11:52:35.642915Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2510.09592","last_updated":"2026-05-10T15:21:35Z","snapshot_observed_at":"2026-07-06T22:32:22.953630Z","submitted_at":"2025-10-10T17:50:59Z","title":"Mind-Paced Speaking: A Dual-Brain Approach to Real-Time Reasoning in Spoken Language Models","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-18T07:43:23.913399Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2510.09592"},"observation_digest":"sha256:2ef3849bc8df65f31a4c08ab888d1c93adca6bd4874a975953c513c199fad1ab","observation_id":"b60aa5e7-3069-4b7e-a59c-a4ba0bbc85f4","resolution":{"observed_at":"2026-05-18T07:46:03.542900Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-04T10:13:42.937202Z","title":"Qian Yang, Jin Xu, Wenrui Liu, Yunfei Chu, Ziyue Jiang, Xiaohuan Zhou, Yichong Leng, Yuanjun Lv, Zhou Zhao, Chang Zhou, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.11098","last_updated":"2026-07-06T08:06:32Z","snapshot_observed_at":"2026-08-04T21:47:58.063486Z","submitted_at":"2025-10-13T07:45:52Z","title":"VCB Bench: An Evaluation Benchmark for Audio-Grounded Large Language Model Conversational Agents","version":5},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T10:13:42.937202Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2510.11098"},"observation_digest":"sha256:37e015ee12c6fa2e04749acd32a2e9337f4b70342cb4f8d17ce8a1df3120c85c","observation_id":"aba706f3-94e5-400c-b97d-419f63c5aace","resolution":{"observed_at":"2026-08-04T10:13:42.937202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2604.15804","last_updated":"2026-04-21T03:35:14Z","snapshot_observed_at":"2026-08-01T22:56:50.755050Z","submitted_at":"2026-04-17T08:05:46Z","title":"Qwen3.5-Omni Technical Report","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-10T08:11:22.402552Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2604.15804"},"observation_digest":"sha256:5718c81470edd16599cc4d91052bf371f65a85acb971d145e6ed732c8a666424","observation_id":"c3c10ec2-4a90-4f64-8d47-f4c4f0029954","resolution":{"observed_at":"2026-05-10T08:12:26.504670Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2605.06765","last_updated":"2026-05-07T17:59:56Z","snapshot_observed_at":"2026-08-02T05:36:23.979419Z","submitted_at":"2026-05-07T17:59:56Z","title":"VITA-QinYu: Expressive Spoken Language Model for Role-Playing and Singing","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-11T01:03:09.942984Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2605.06765"},"observation_digest":"sha256:b0365f85d3ad0397ab2af903f3957429ea506ce8c1a5f502bb7aa466c3b69f3f","observation_id":"5be8038e-bcc9-4550-ba44-09d2b089ce63","resolution":{"observed_at":"2026-05-11T04:50:56.186750Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2605.09413","last_updated":"2026-05-10T08:28:58Z","snapshot_observed_at":"2026-07-06T23:21:30.897592Z","submitted_at":"2026-05-10T08:28:58Z","title":"Evaluating the Expressive Appropriateness of Speech in Rich Contexts","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-12T01:56:11.473572Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2605.09413"},"observation_digest":"sha256:33657c597b84217ead9989cab65d76dcf1157dc81ba6e5528547f81d76825b13","observation_id":"cab538ae-c1ca-426a-bfce-4a2d31dfbdcf","resolution":{"observed_at":"2026-05-12T01:56:14.573707Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2605.20266","last_updated":"2026-05-18T20:21:32Z","snapshot_observed_at":"2026-08-03T05:12:45.221230Z","submitted_at":"2026-05-18T20:21:32Z","title":"A Survey of Large Audio Language Models: Generalization, Trustworthiness, and Outlook","version":1},"reference_index":181,"source":"pdf_text","source_observed_at":"2026-05-21T07:38:23.099479Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2605.20266"},"observation_digest":"sha256:e730a20fbd37d034abfaca5d2f2ffe08802432bc88bf988af7a9acecada4333a","observation_id":"47f6c4d8-2d33-41ab-8c32-a3db3e5085b2","resolution":{"observed_at":"2026-05-21T07:39:49.047504Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2605.20755","last_updated":"2026-06-11T05:13:51Z","snapshot_observed_at":"2026-08-02T13:39:32.672112Z","submitted_at":"2026-05-20T05:54:08Z","title":"DuplexSLA: A Full-Duplex Spoken Language Model with Synchronized Speech, Language, and Action","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-21T02:41:13.583493Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2605.20755"},"observation_digest":"sha256:b45d15f5a86be1ec599e1646937b5cb8371eb3a7c084f78c53b67e204855ef7b","observation_id":"330ba7bf-eb71-4b78-873f-bddb8994d002","resolution":{"observed_at":"2026-05-21T02:43:55.060576Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2605.20755","last_updated":"2026-06-11T05:13:51Z","snapshot_observed_at":"2026-08-02T13:39:32.672112Z","submitted_at":"2026-05-20T05:54:08Z","title":"DuplexSLA: A Full-Duplex Spoken Language Model with Synchronized Speech, Language, and Action","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-30T17:32:58.848455Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2605.20755"},"observation_digest":"sha256:61641f706054bdf1c050324dfcf28758ffc326a1eb8a1de94e11d7740d36326a","observation_id":"d981c1f0-b2bf-43c5-9c63-56797b1abc01","resolution":{"observed_at":"2026-06-30T17:34:57.301296Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2605.21008","last_updated":"2026-05-20T10:44:56Z","snapshot_observed_at":"2026-07-06T23:31:32.577242Z","submitted_at":"2026-05-20T10:44:56Z","title":"A Survey of Audio Reasoning in Multimodal Foundation Models","version":1},"reference_index":129,"source":"pdf_text","source_observed_at":"2026-05-21T02:08:06.976461Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2605.21008"},"observation_digest":"sha256:22151610b4f8ad2b4a4fa9eb003e24428284228943b7d523a310d34cd84f6df5","observation_id":"bdd1c086-e010-41c2-8fd5-4207109c55dd","resolution":{"observed_at":"2026-05-21T02:09:24.437109Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2606.01016","last_updated":"2026-05-31T05:13:32Z","snapshot_observed_at":"2026-07-06T23:41:39.172310Z","submitted_at":"2026-05-31T05:13:32Z","title":"PolySpeech-100: A Large-Scale Benchmark for Speech Understanding Across 100+ Languages and Dialects","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-06-28T17:44:07.669223Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2606.01016"},"observation_digest":"sha256:a60c48c091b72b18c9d85363e75a1eff427f952961c6ae384b6c47754507241f","observation_id":"4cad0548-0107-41e4-a95f-f154e1f5ac0e","resolution":{"observed_at":"2026-06-28T17:52:27.009937Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2606.05896","last_updated":"2026-06-22T18:29:42Z","snapshot_observed_at":"2026-07-06T23:45:51.910263Z","submitted_at":"2026-06-04T09:03:43Z","title":"Resonant Minds: Closed-Loop Social Avatars with Theory of Mind","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-28T02:04:39.753443Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2606.05896"},"observation_digest":"sha256:6bcf785ac7435f6ec021cef5d38dbd087f80dcf94526a9da5e7d21e7413cd6a9","observation_id":"b94e9da7-36be-463a-aa74-4ad1a086ea3d","resolution":{"observed_at":"2026-07-02T12:36:56.235234Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2606.07547","last_updated":"2026-05-04T17:54:41Z","snapshot_observed_at":"2026-08-07T09:08:57.491441Z","submitted_at":"2026-05-04T17:54:41Z","title":"Liberating LLM Capabilities in Full-Duplex Speech Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-01T00:13:14.701980Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2606.07547"},"observation_digest":"sha256:6d888a666c787d469691d1b62e77d648b23e1ddcc80448be86574e7c393018c2","observation_id":"30a4e08a-3965-4cca-931a-e968a2deb774","resolution":{"observed_at":"2026-07-01T00:15:09.030108Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-01T11:20:18.079021Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19932","last_updated":"2026-07-22T09:04:55Z","snapshot_observed_at":"2026-08-07T17:53:01.414682Z","submitted_at":"2026-07-22T09:04:55Z","title":"Efficient Chain-of-Modality Reasoning via Progressive Compression for Spoken Language Models","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-01T11:20:18.079021Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2607.19932"},"observation_digest":"sha256:4ac7469438637b39d814a12d17d4f8c9fc17ff78ad94dcde4dbd2f1b4c8d193b","observation_id":"533cc13a-9506-4390-bb4e-c9a1d0468fef","resolution":{"observed_at":"2026-08-01T11:20:18.079021Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.17810/citation-record","integrity":"/paper/2502.17810/integrity","json":"/paper/2502.17810/citation-record.json","paper":"/paper/2502.17810"},"outbound":[],"paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 17 inbound Pith citation observations for arXiv:2502.17810."}