{"as_of":"2026-08-13T21:45:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2a75e621454a4002bf9958b53cc486eba8c4d802b00d09b68b18fc6e76194d64","coverage":[{"denominator":29,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":29,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T15:22:02.107863Z","state":"measured"},{"denominator":30,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":30,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T16:37:24.214630Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.09186","snapshot_observed_at":"2026-08-01T16:37:24.214630Z","title":"Huang et al","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.26410","last_updated":"2026-07-29T02:42:56Z","snapshot_observed_at":"2026-08-03T19:09:06.764457Z","submitted_at":"2026-07-29T02:42:56Z","title":"Voice Memory for Agentic Speech Recognition","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-01T16:37:24.214630Z"},"links":{"cited_paper":"/paper/2606.09186","citing_paper":"/paper/2607.26410"},"observation_digest":"sha256:becc692e42c3132e7956875f3a4619d3e5f522f455895753b4a66269f7e1faa3","observation_id":"89b5d4f3-5ebc-40ac-b54e-9fcb33ce1ef0","resolution":{"observed_at":"2026-08-01T16:37:24.214630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2606.09186/citation-record","integrity":"/paper/2606.09186/integrity","json":"/paper/2606.09186/citation-record.json","paper":"/paper/2606.09186"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.06502","last_updated":"2025-09-08T10:07:30Z","snapshot_observed_at":"2026-08-12T21:09:54.215392Z","submitted_at":"2025-09-08T10:07:30Z","title":"FireRedChat: A Pluggable, Full-Duplex Voice Interaction System with Cascaded and Semi-Cascaded Implementations","version":1},"cited_work":{"arxiv_id":"2509.06502","doi":"10.48550/arxiv.2509.06502","metadata_source":"pith","pith_arxiv_id":"2509.06502","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Fireredchat: A pluggable, full-duplex voice in- teraction system with cascaded and semi-cascaded implementa- tions","venue":"cs.SD","work_id":"373dad83-3519-4449-a200-f05a328b52d1","year":2025},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2509.06502","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:aef94c1207bb00a1b6c096e3feb0f575935b18a7c653878843a32164d2ab045f","observation_id":"7de6918a-8e30-45ad-bd8b-ba362a98e847","resolution":{"observed_at":"2026-07-03T03:27:34.977710Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:18:48.272754+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:18:48.272754+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10759","last_updated":"2024-07-15T14:38:09Z","snapshot_observed_at":"2026-08-13T05:33:22.244747Z","submitted_at":"2024-07-15T14:38:09Z","title":"Qwen2-Audio Technical Report","version":1},"cited_work":{"arxiv_id":"2407.10759","doi":"10.48550/arxiv.2407.10759","metadata_source":"pith","pith_arxiv_id":"2407.10759","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2-Audio Technical Report","venue":"eess.AS","work_id":"c249e63c-cf40-408f-a4ff-fdf68e8cbeb8","year":2024},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2407.10759","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:4832de3551f6d4588478432ae14afb8780caf43b830c974490f1a70c8b9e937a","observation_id":"cd1e3c9b-e5eb-41aa-9191-8d12d5f9c043","resolution":{"observed_at":"2026-07-03T03:27:34.967180Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00037","last_updated":"2024-10-02T09:11:45Z","snapshot_observed_at":"2026-07-30T10:21:14.474746Z","submitted_at":"2024-09-17T17:55:39Z","title":"Moshi: a speech-text foundation model for real-time dialogue","version":2},"cited_work":{"arxiv_id":"2410.00037","doi":"10.1121/1.1906946","metadata_source":"pith","pith_arxiv_id":"2410.00037","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Moshi: a speech-text foundation model for real-time dialogue","venue":"eess.AS","work_id":"3104332b-d279-44c8-aaa7-3d5a13c01832","year":2024},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2410.00037","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:2be4bd06d50a6f5ace1ccf17d486c269195606a5c4f27f1c505d288603ba810f","observation_id":"6c6575e6-57dd-432a-bd55-12a5ff1fc009","resolution":{"observed_at":"2026-07-03T03:27:34.974880Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14233","last_updated":"2023-05-23T16:49:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-23T16:49:14Z","title":"Enhancing Chat Language Models by Scaling High-quality Instructional Conversations","version":1},"cited_work":{"arxiv_id":"2305.14233","doi":"10.48550/arxiv.2305.14233","metadata_source":"pith","pith_arxiv_id":"2305.14233","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Enhancing Chat Language Models by Scaling High-quality Instructional Conversations","venue":"cs.CL","work_id":"5b264119-de53-49d1-b7be-d172bf51c834","year":2023},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2305.14233","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:57ed51c31d58da65a27829b7b1318f5c12308575cfeeb58fda4e7923b5028fb5","observation_id":"616e258a-1439-478f-b0b9-83aa4e3fe9a0","resolution":{"observed_at":"2026-07-03T03:27:34.990361Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-05-22T21:23:27.379194+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-22T21:23:27.379194+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T15:22:02.107863Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:e49f74cc499d726077f35314f74240134e80b3b4083bd22c462abc1bce74103e","observation_id":"98fb4e5e-9dad-49dd-88fa-f8daedc1774a","resolution":{"observed_at":"2026-06-27T15:22:02.107863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2501.15368","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T16:49:57.284405Z","title":"Baichuan-omni-1.5 technical report","venue":null,"work_id":"d2c9e57e-0e5b-4999-b035-ef159319e83d","year":2025},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:f5c5a1514cb1c10cba7c0899c3f0b87518d04a7b7b0c0866d02ca79b574a704a","observation_id":"47e44b60-ac6b-4a55-9448-b864553eae8e","resolution":{"observed_at":"2026-07-03T03:27:34.948374Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07854","last_updated":"2023-04-16T18:37:39Z","snapshot_observed_at":"2026-08-13T17:51:01.412646Z","submitted_at":"2023-04-16T18:37:39Z","title":"Towards Better Instruction Following Language Models for Chinese: Investigating the Impact of Training Data and Evaluation","version":1},"cited_work":{"arxiv_id":"2304.07854","doi":"10.48550/arxiv.2304.07854","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.07854","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Towards better instruction following language models for chinese: Investigating the im- pact of training data and evaluation","venue":"arXiv (Cornell University)","work_id":"58a28446-45c2-475f-80c3-eb1b2f442d58","year":2023},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2304.07854","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:7b06861b728bf69a766efbe92349f329a0b131a4a15a62a22ff52041e2a9f02d","observation_id":"f3ebd363-802d-46ed-b0c1-22820ed014bf","resolution":{"observed_at":"2026-07-03T03:27:34.996815Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.18425","last_updated":"2025-04-25T15:31:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-25T15:31:46Z","title":"Kimi-Audio Technical Report","version":1},"cited_work":{"arxiv_id":"2504.18425","doi":"10.48550/arxiv.2504.18425","metadata_source":"pith","pith_arxiv_id":"2504.18425","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Kimi-Audio Technical Report","venue":"eess.AS","work_id":"9c2ba56b-5585-4f28-b751-703f31dca2d5","year":2025},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2504.18425","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:ff9f41292b522cbd1b6499424e48e3ac37b803e0d9d308e02b10ebe19ddc54d2","observation_id":"520d023e-46ba-4a4d-9dc3-fb488f1c84ce","resolution":{"observed_at":"2026-07-03T03:27:34.934034Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07327","last_updated":"2023-10-31T11:38:07Z","snapshot_observed_at":"2026-08-13T12:02:41.563345Z","submitted_at":"2023-04-14T18:01:29Z","title":"OpenAssistant Conversations -- Democratizing Large Language Model Alignment","version":2},"cited_work":{"arxiv_id":"2304.07327","doi":"10.48550/arxiv.2304.07327","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.07327","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"R., Stevens, K., Barhoum, A., Duc, N","venue":"arXiv (Cornell University)","work_id":"34ddaebd-853b-4596-b47d-a42b9340305e","year":2024},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2304.07327","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:97d6a303baf3cafe26864e1b8ccacd2558ceef97f359f148ca633fd1fad77320","observation_id":"a3aada79-a49b-4af6-b4b1-98a1a95fe510","resolution":{"observed_at":"2026-07-03T03:27:34.964427Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T15:22:02.107863Z","title":"InIEEE Automatic Speech Recognition and Understanding Workshop, ASRU 2025, Honolulu, HI, USA, December 6-10, 2025, pages 1–8","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:bd548960a24efaade932b628ec21d0cbe5cc0054449ddac413731c11839c1c90","observation_id":"95185f7e-3b6e-4b9c-8acf-470044852c07","resolution":{"observed_at":"2026-06-27T15:22:02.107863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01268","last_updated":"2025-06-02T02:40:46Z","snapshot_observed_at":"2026-08-12T20:35:57.955610Z","submitted_at":"2025-06-02T02:40:46Z","title":"CleanS2S: Single-file Framework for Proactive Speech-to-Speech Interaction","version":1},"cited_work":{"arxiv_id":"2506.01268","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01268","snapshot_observed_at":"2026-07-03T03:27:34.992590Z","title":"Ziyang Ma, Yakun Song, Chenpeng Du, Jian Cong, Zhuo Chen, Yuping Wang, Yuxuan Wang, and Xie Chen","venue":null,"work_id":"a44294b6-6149-45bf-a7cb-e22bee426343","year":null},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2506.01268","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:737d7f59851969c09de3e832fc891305ab895688a504f8f9e953e407df571785","observation_id":"da5bd0c2-fcbc-43ca-b841-79eaff1e1b73","resolution":{"observed_at":"2026-07-03T03:27:34.994077Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T15:22:02.107863Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:9d68730183298f4edfa674ea8d9aa985be371d7945a039631b3929d50ae2cbee","observation_id":"719b2b1d-0a84-4af9-8df2-24f3c64f0019","resolution":{"observed_at":"2026-06-27T15:22:02.107863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.16502","last_updated":"2022-11-22T19:20:29Z","snapshot_observed_at":"2026-08-13T16:12:20.346226Z","submitted_at":"2022-03-30T17:39:45Z","title":"Generative Spoken Dialogue Language Modeling","version":2},"cited_work":{"arxiv_id":"2203.16502","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2203.16502","snapshot_observed_at":"2026-07-03T03:27:34.965558Z","title":"Generative spoken dialogue language modeling","venue":null,"work_id":"e42ddcbf-ecec-43e8-a4bd-3ef9c0029c0f","year":null},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2203.16502","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:3a5a929390bb66df4c29c54a0919fd99cf9a2739514f76fef7fab20629d44265","observation_id":"1b0a8841-f964-46db-991b-6d8ad15baee3","resolution":{"observed_at":"2026-07-03T03:27:34.967134Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":"2410.21276","doi":"10.1177/15248380231178756","metadata_source":"pith","pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4o System Card","venue":"cs.CL","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","year":2024},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:4498da461f6398e421fa73fdaa6f1da4d2167cc73225c0bc79c17efd9f814632","observation_id":"1424cab8-256d-4987-901d-2e21e602c5b8","resolution":{"observed_at":"2026-07-03T03:27:34.925449Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T15:22:02.107863Z","title":"In 2015 IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2015, South Brisbane, Queensland, Australia, April 19-24, 2015, pages 5206–5210","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:f1808953ebca5701a677e05a461c05f791ba4892605ab06fa19e902aebb544e9","observation_id":"f486d2a1-263a-4b1a-b8bd-ad83515765f5","resolution":{"observed_at":"2026-06-27T15:22:02.107863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":"2403.05530","doi":"10.48550/arxiv.2403.05530","metadata_source":"pith","pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":"cs.CL","work_id":"80e3e977-f1bb-4c83-8d0c-1ab0a0c5c3f1","year":2024},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:a08de9a6531065fc33f69777b2d9d9258f8641ac1b0f773635fc62e475a72ba9","observation_id":"1540ff3a-9419-411e-b956-0665f08be33c","resolution":{"observed_at":"2026-07-03T03:27:34.951605Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04615","last_updated":"2023-06-12T17:51:15Z","snapshot_observed_at":"2026-07-06T13:19:12.109592Z","submitted_at":"2022-06-09T17:05:34Z","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","version":3},"cited_work":{"arxiv_id":"2206.04615","doi":"10.1162/tacl_a_00688","metadata_source":"pith","pith_arxiv_id":"2206.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","venue":"cs.CL","work_id":"bb63abb3-0d50-4362-b97c-b5e725b03b39","year":2022},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2206.04615","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:021910a1e18b76689efd46cf4dea3db16da7167e00b882e621d6a1613a4975e5","observation_id":"4c428786-99d2-4a8a-a407-84414cd2cada","resolution":{"observed_at":"2026-07-03T03:27:34.984043Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.09261","last_updated":"2022-10-17T17:08:26Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-10-17T17:08:26Z","title":"Challenging BIG-Bench Tasks and Whether Chain-of-Thought Can Solve Them","version":1},"cited_work":{"arxiv_id":"2210.09261","doi":"10.48550/arxiv.2210.09261","metadata_source":"pith","pith_arxiv_id":"2210.09261","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Challenging BIG-Bench Tasks and Whether Chain-of-Thought Can Solve Them","venue":"cs.CL","work_id":"513eb205-04ca-4722-9a43-a74e8cbe7e85","year":2022},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2210.09261","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:ecfe19f837bd9aada2906a3ae2ff9d707c1b4579030e3fa67b089555bf4ee90b","observation_id":"45001c07-ce4d-46b9-982f-cc7b3b081eaa","resolution":{"observed_at":"2026-07-03T03:27:34.933850Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:ac4573087d1929c3c990ba4ffb0478d59d0f3e1804b8b238fbcd77dee46bb9ac","observation_id":"b1ba49d5-88d7-4bb1-a437-86645d3b9ac5","resolution":{"observed_at":"2026-07-03T03:27:34.964777Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.20156","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T13:59:52.841947Z","title":"Fun-audio-chat technical report","venue":null,"work_id":"3aab8cff-d938-401b-a33e-ed1821101046","year":2025},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:89de18a4897eea1063a7587c67f5a2d52045180849006e25bc7409146329fa97","observation_id":"6075f2df-37ec-4ec1-9a62-71f21f480513","resolution":{"observed_at":"2026-07-03T03:27:34.982969Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19487","last_updated":"2024-10-29T17:44:03Z","snapshot_observed_at":"2026-08-12T23:54:48.382730Z","submitted_at":"2024-05-29T20:05:46Z","title":"A Full-duplex Speech Dialogue Scheme Based On Large Language Models","version":2},"cited_work":{"arxiv_id":"2405.19487","doi":"10.48550/arxiv.2405.19487","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.19487","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A full-duplex speech dialogue scheme based on large language model","venue":"arXiv (Cornell University)","work_id":"ad2d6e58-10ce-411e-8c42-69599642b323","year":2024},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2405.19487","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:158f7eb24f85c086656d4a8ec58c8369ba6e69311f4904e0177bd035ed9b2b7e","observation_id":"4fbddfca-2ba8-4d45-9932-5fc094535bec","resolution":{"observed_at":"2026-07-03T03:27:34.976859Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.23808","doi":"10.48550/arxiv.2512.23808","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mimo-audio: Audio language models are few-shot learners","venue":"arXiv (Cornell University)","work_id":"8c438dfc-e0fb-45e5-983b-3708cd7afa91","year":2025},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:88e98a671b9b1a850f90c5ad58deaed81d35b33577b2d8e556568635997178c5","observation_id":"677c4174-76b2-4d78-a095-13dd3c96d46b","resolution":{"observed_at":"2026-07-03T03:27:34.987343Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.15827","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T03:27:34.978290Z","title":"Mini-omni-reasoner: Token-level thinking-in-speaking in large speech models","venue":null,"work_id":"42d472a8-566e-410c-b5ec-6b3df001dfd5","year":2025},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:ad185fca4c2155c7ed6b66b64813f124d86434a0ecd6b191b751d24afa35ddb4","observation_id":"6a7e3871-3f95-4a75-81f1-ed871e2981fe","resolution":{"observed_at":"2026-07-03T03:27:34.980125Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20215","last_updated":"2025-03-26T04:17:55Z","snapshot_observed_at":"2026-08-06T08:46:20.194739Z","submitted_at":"2025-03-26T04:17:55Z","title":"Qwen2.5-Omni Technical Report","version":1},"cited_work":{"arxiv_id":"2503.20215","doi":"10.48550/arxiv.2503.20215","metadata_source":"pith","pith_arxiv_id":"2503.20215","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5-Omni Technical Report","venue":"cs.CL","work_id":"438f105c-fa9b-44aa-ad52-43acb8045cda","year":2025},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2503.20215","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:dc1a5e2f713bd39a60ec5f73b9b8e6c53d49c559e37059abbbcf0a1588ba81ed","observation_id":"1c71c09e-45b4-4f30-ad3c-be4af4142bce","resolution":{"observed_at":"2026-07-03T03:27:34.942903Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:18:44.887499+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:18:44.887499+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.09180","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T08:29:42.174877Z","title":"Duplexcascade: Full- duplex speech-to-speech dialogue with vad-free cascaded asr- llm-tts pipeline and micro-turn optimization,","venue":null,"work_id":"26a7bff2-04ba-4f82-a492-a29691343810","year":2026},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:a11f46cc33cd2000738ea3128042b7d500fc31295933e5a330799800e7791858","observation_id":"5a806667-89b4-40b5-ab16-1e6a32e8592b","resolution":{"observed_at":"2026-07-03T03:27:34.980639Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.01800","last_updated":"2024-08-03T15:02:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-03T15:02:21Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","version":1},"cited_work":{"arxiv_id":"2408.01800","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.01800","snapshot_observed_at":"2026-07-10T11:37:03.161139Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","venue":"cs.CV","work_id":"0f06e436-0c76-4e3c-be5e-6168f6bc4336","year":2024},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2408.01800","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:b93590d556a3ac01c225aeb8bc113ca018224582105e4b7a3fc7cdd844d0bc9d","observation_id":"830949d4-e30d-4e7b-b065-6f7a62fae384","resolution":{"observed_at":"2026-07-03T03:27:34.973705Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.17060","last_updated":"2025-05-17T08:13:59Z","snapshot_observed_at":"2026-08-10T23:13:14.879038Z","submitted_at":"2025-05-17T08:13:59Z","title":"SALMONN-omni: A Standalone Speech LLM without Codec Injection for Full-duplex Conversation","version":1},"cited_work":{"arxiv_id":"2505.17060","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.17060","snapshot_observed_at":"2026-07-03T16:28:39.133843Z","title":"Salmonn-omni: A standalone speech llm without codec injection for full- duplex conversation.arXiv preprint arXiv:2505.17060","venue":null,"work_id":"5b053464-9fd0-4025-a38f-5c6ae551187c","year":2025},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2505.17060","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:7fe6422796f711d8bdf935f43fcf367087f23e9d1196b1413cb1021fbcc16fad","observation_id":"030fe35c-2674-446b-b5c0-b3fc2ffab60a","resolution":{"observed_at":"2026-07-03T03:27:34.936721Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.01470","last_updated":"2024-05-02T17:00:02Z","snapshot_observed_at":"2026-08-13T13:43:58.816757Z","submitted_at":"2024-05-02T17:00:02Z","title":"WildChat: 1M ChatGPT Interaction Logs in the Wild","version":1},"cited_work":{"arxiv_id":"2405.01470","doi":"10.48550/arxiv.2405.01470","metadata_source":"pith","pith_arxiv_id":"2405.01470","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WildChat: 1M ChatGPT Interaction Logs in the Wild","venue":"cs.CL","work_id":"799d0d7b-7d66-40bb-b17f-205e5d2f3e13","year":2024},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"cited_paper":"/paper/2405.01470","citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:a59517ef018dfe8cf7f711647cc900f8599fd32d40307433a0875baf85776ec4","observation_id":"e301429a-ae85-490a-b8ac-a531e9cb089e","resolution":{"observed_at":"2026-07-03T03:27:34.931397Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.17862","doi":"10.48550/arxiv.2505.17862","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlvu: Benchmarking multi-task long video understanding","venue":"arXiv (Cornell University)","work_id":"8c7aec87-020f-4959-9199-7df8b9231cc4","year":2025},"citing_paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:02.107863Z"},"links":{"citing_paper":"/paper/2606.09186"},"observation_digest":"sha256:778558509371f51e5c0de24635ca7a02852e0ec502849f476539cd61f1c679f5","observation_id":"569fe23b-d8b0-46b2-9cab-4d122910491d","resolution":{"observed_at":"2026-07-03T03:27:34.970446Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2606.09186","last_updated":"2026-06-08T08:23:02Z","latest_version":1,"primary_category":"cs.HC","snapshot_observed_at":"2026-08-12T23:18:48.473844Z","submitted_at":"2026-06-08T08:23:02Z","title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction"},"reference_resolution":{"displayed":29,"state_counts":{"malformed_identifier":1,"metadata_mismatch":16,"parse_uncertain":0,"unresolved":4,"verified_exact":8,"verified_fuzzy":0},"total_outbound_references":29},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 29 of 29 outbound references and 1 inbound Pith citation observation for arXiv:2606.09186."}