{"as_of":"2026-08-06T12:20:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:70760a6f1e3609c05cfa5e8f21e6910b1bdaf75f30426c84babf3bd8a0a1803f","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T18:22:08.670559Z","state":"measured"},{"denominator":40,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":40,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-31T15:38:33.184463Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-05-11T00:41:07.116524Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"cited_work":{"arxiv_id":"2604.09121","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.09121","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","venue":"cs.CL","work_id":"d9c5a81f-6f9a-4a14-92a0-9696c92a150c","year":2026},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/2604.09121","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:0a06c07ed6b23f155d5ab73fa33ea1720f25e3fd811a0336455432e1d9332923","observation_id":"81085b5b-0f18-4005-bc94-bd24329ffaf0","resolution":{"observed_at":"2026-05-11T00:41:07.124093Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.09121","snapshot_observed_at":"2026-07-31T15:38:33.184463Z","title":"arXiv:2604.09121","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2607.28175","last_updated":"2026-07-30T13:12:25Z","snapshot_observed_at":"2026-08-03T04:52:18.523406Z","submitted_at":"2026-07-30T13:12:25Z","title":"AgenticASR: Refining Speech Recognition in Real-World Scenarios via an Agentic Approach","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-31T15:38:33.184463Z"},"links":{"cited_paper":"/paper/2604.09121","citing_paper":"/paper/2607.28175"},"observation_digest":"sha256:3c75396a643a43df5eee61a6c29df2cdf2726dda5c9cadc7c252ca92ad9f889d","observation_id":"5fd54434-8aa9-4773-8c64-6953c2117224","resolution":{"observed_at":"2026-07-31T15:38:33.184463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2604.09121/citation-record","integrity":"/paper/2604.09121/integrity","json":"/paper/2604.09121/citation-record.json","paper":"/paper/2604.09121"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"cited_work":{"arxiv_id":"2604.09121","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.09121","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","venue":"cs.CL","work_id":"d9c5a81f-6f9a-4a14-92a0-9696c92a150c","year":2026},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/2604.09121","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:0a06c07ed6b23f155d5ab73fa33ea1720f25e3fd811a0336455432e1d9332923","observation_id":"81085b5b-0f18-4005-bc94-bd24329ffaf0","resolution":{"observed_at":"2026-05-11T00:41:07.124093Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"To address this, several semantic-aware metrics have been proposed","venue":null,"work_id":"db7c359b-47dd-4c4a-a683-208e2581a84f","year":null},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:b333e5853adacba66f870f537a4692f94a49189c2c1d7607cf362258983430c0","observation_id":"f510ac35-51de-4ad0-a6f9-68bca64dc505","resolution":{"observed_at":"2026-05-17T04:09:00.654357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Moving to embedding-based evaluation, SemDist [18] utilized RoBERTa-based sentence embeddings to measure se- mantic similarity beyond literal overlap","venue":null,"work_id":"f52ce83e-95da-4bc0-ac21-fe1577993b56","year":null},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:9136dd588f767d45d21255d7180f54366c87257046ff91eb0cc8e3081fb738e3","observation_id":"0c56680e-565e-4ae3-8240-accc6d238499","resolution":{"observed_at":"2026-05-17T04:09:00.645572Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Unlike these continuous scoring metrics, our LLM-as-a-Judge adopts a binary functional criterion, acting as a strict gatekeeper to determine if the user’s intent is executable","venue":null,"work_id":"240d57d4-c8ba-4de1-854c-6bbfd610a5d3","year":null},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:a7a5dd83575788b22c87ef8c226ff0315a3763bb124938b9a0ab135daac1b69b","observation_id":"2bc9f8e2-84c4-4fcb-a24d-67b8fcedaec9","resolution":{"observed_at":"2026-05-17T04:09:00.656952Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e1a07c9b-25be-4058-bfff-65d5e424adb0","year":null},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:02ec34e35f749023689a89fa90eeadf2e1bf377773af9c57a9adfc9e29c91f88","observation_id":"c95d515a-2140-41e0-821c-7a22e0817e42","resolution":{"observed_at":"2026-05-17T04:09:00.651914Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"55f2f0f5-3a4e-4004-9094-a4f23f0aed5f","year":null},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:d3f5f3e023db3965c884a649eec51b57ef5cd0bc37d2746203f0cdffa2869ec0","observation_id":"6ea06b30-0dd1-4bd1-8372-af3af732518c","resolution":{"observed_at":"2026-05-17T04:09:00.618835Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"We first outline the experimental setup in Section 5.1, detailing the diverse benchmarks and the foun- dational models","venue":null,"work_id":"4f585b7d-0fe1-4a4f-bfbc-1f1aadd5c997","year":null},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:7b0646f3602fb6ae343fed11c7f4724aa6c47a019d15a672cfcc72a0aa2f6b1e","observation_id":"9471a8d1-4bbf-48f6-88d3-41290371933e","resolution":{"observed_at":"2026-05-17T04:09:00.629267Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"We introduced 𝑆2𝐸 𝑅, a novel met- ric that leverages LLMs as judges to prioritize sentence-level semantic coherence","venue":null,"work_id":"bdebea48-96e0-419a-9c13-56a396e8dd7e","year":null},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:ed71b1bf32fac1af9ac877b042c36b650f929c1298a81394bc317e8a841cf7ec","observation_id":"06165007-8c03-4914-86e4-4c039d9b7e2d","resolution":{"observed_at":"2026-05-17T04:09:00.611309Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Con- nectionist temporal classification: labelling unsegmented se- quence data with recurrent neural networks","venue":null,"work_id":"ab024f78-658b-4fa9-886c-6cad36f4af9d","year":2006},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:dfedc970964d567b48f6247bc618024bc6f88e0d381c35aa5d3d74e9d22c65e8","observation_id":"5747ccab-8640-46d0-8aef-42b29be4fc84","resolution":{"observed_at":"2026-05-17T04:09:00.603920Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1211.3711","last_updated":"2012-11-14T19:25:21Z","snapshot_observed_at":"2026-07-06T02:59:58.238741Z","submitted_at":"2012-11-14T19:25:21Z","title":"Sequence Transduction with Recurrent Neural Networks","version":1},"cited_work":{"arxiv_id":"1211.3711","doi":"10.48550/arxiv.1211.3711","metadata_source":"pith","pith_arxiv_id":"1211.3711","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sequence Transduction with Recurrent Neural Networks","venue":"cs.NE","work_id":"4553c1f9-9627-48f1-bcb4-69ce67d2ad80","year":2012},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/1211.3711","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:a1b3c393f847ba3adc169647f3c5432ac25f8c84304ce2d1c0397dc7295886e4","observation_id":"7dc7159d-1057-4c8d-b02f-824e006cbb15","resolution":{"observed_at":"2026-05-11T00:41:07.208345Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Listen, attend and spell: A neural network for large vocabulary conversational speech recognition","venue":null,"work_id":"8f70e30a-810e-4ef7-8ecc-7671ec9c9a97","year":2016},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:bde9fa98e04c9fa100ee69be294464538dc348fb0360e0098519e9aa8e1a4a13","observation_id":"4ca584d7-84f7-49e0-9a6a-a5b30e85a17e","resolution":{"observed_at":"2026-05-17T04:09:00.606507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T08:34:49.851530Z","title":"Robust speech recognition via large-scale weak supervision","venue":null,"work_id":"ccda886b-9b8b-445f-a7a7-6584c21d61cc","year":2023},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:1ae06505db51f1147b5df3a03cb255bbead3e141b649f00fd43bed2362bedbda","observation_id":"1b505514-3f76-4160-9476-f549f29ba05c","resolution":{"observed_at":"2026-05-17T04:09:00.608758Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"On decoder-only architecture for speech- to-text and large language model integration","venue":null,"work_id":"52ed6a01-c8e9-44e0-87a8-674431586d9e","year":2023},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:64535f8033d04ad8a97b79ba6e64c4a6ae3c25fb256d1b701ab7e84661dd9d5f","observation_id":"6dd10421-50b6-415c-9de0-cdc2836f52aa","resolution":{"observed_at":"2026-05-17T04:09:00.613775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Slm: Bridge the thin gap between speech and text foundation models","venue":null,"work_id":"f21dddb2-28ed-4f80-88d0-6299c7a6ba06","year":2023},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:54859aa74fe5343102e55000d372cd3fbf616580ec20306a5ffd2244d03687b0","observation_id":"712c8ab5-c4bb-4aca-8bb0-ea618194b3f2","resolution":{"observed_at":"2026-05-17T04:09:00.621327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13289","last_updated":"2024-04-08T06:12:52Z","snapshot_observed_at":"2026-08-02T21:51:36.809095Z","submitted_at":"2023-10-20T05:41:57Z","title":"SALMONN: Towards Generic Hearing Abilities for Large Language Models","version":2},"cited_work":{"arxiv_id":"2310.13289","doi":"10.1016/j.ajhg.2009.06.010","metadata_source":"pith","pith_arxiv_id":"2310.13289","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SALMONN: Towards Generic Hearing Abilities for Large Language Models","venue":"cs.SD","work_id":"b06d6def-ea7a-4cbd-8bb3-73f6a81db238","year":2023},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/2310.13289","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:1d7a0ab5a84cc91bde1e706e6cfb28d4b01d8649affcaa81edd324b8e2c185de","observation_id":"33fe5cca-c2bf-43be-b7ab-f37daaebd684","resolution":{"observed_at":"2026-05-18T02:29:46.683133Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07919","last_updated":"2023-12-21T10:20:42Z","snapshot_observed_at":"2026-07-06T16:47:12.709738Z","submitted_at":"2023-11-14T05:34:50Z","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","version":2},"cited_work":{"arxiv_id":"2311.07919","doi":"10.48550/arxiv.2311.07919","metadata_source":"pith","pith_arxiv_id":"2311.07919","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","venue":"eess.AS","work_id":"d3f033ac-bfa8-4143-9d0d-51f3f5bd3f0e","year":2023},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/2311.07919","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:87c0413259d2d747b4128ca321a27a805936a8dad7494bf16a334ff2dc82714f","observation_id":"290cb93c-7d3d-4bcb-bfbd-7e3b0930fc6b","resolution":{"observed_at":"2026-05-12T18:57:28.886773Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Prompt- ing large language models with speech recognition abilities","venue":null,"work_id":"595e2e2a-cb62-4302-9e40-4d8e8a3235cf","year":2024},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:b7b403258837ae8618ab7928c08b10faa04fde20090d43ce47e57df012860e10","observation_id":"ffd6bc38-65c5-46f9-8560-d3ba1fb4f76d","resolution":{"observed_at":"2026-05-17T04:09:00.642980Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08846","last_updated":"2024-02-13T23:25:04Z","snapshot_observed_at":"2026-08-06T09:11:01.697195Z","submitted_at":"2024-02-13T23:25:04Z","title":"An Embarrassingly Simple Approach for LLM with Strong ASR Capacity","version":1},"cited_work":{"arxiv_id":"2402.08846","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.08846","snapshot_observed_at":"2026-07-10T20:37:34.440026Z","title":"An embarrassingly simple approach for LLM with strong ASR capacity","venue":"cs.CL","work_id":"d0cf5313-bc88-4f8f-9a2c-9012a25a93dd","year":2024},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/2402.08846","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:7f67cee7a6e09e956b9a2ffa4c114c94ddfc609c146051de71dae3562ad157a2","observation_id":"63ac30c2-c9f2-4acb-97ca-3450bc6c544e","resolution":{"observed_at":"2026-05-11T00:45:49.116917Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":"2407.04675","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-07-05T13:21:06.342183Z","title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","venue":"eess.AS","work_id":"c5c60033-9068-454d-8df1-52efb011f98b","year":2024},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:b11726a53f7be4b130db347df8873ded3385c2071d2a3ea15ce80d0032f764af","observation_id":"0bae1a1b-aacb-4176-a926-4b627897d09d","resolution":{"observed_at":"2026-05-11T00:41:07.223921Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14350","last_updated":"2025-01-24T09:21:41Z","snapshot_observed_at":"2026-08-05T01:17:13.454546Z","submitted_at":"2025-01-24T09:21:41Z","title":"FireRedASR: Open-Source Industrial-Grade Mandarin Speech Recognition Models from Encoder-Decoder to LLM Integration","version":1},"cited_work":{"arxiv_id":"2501.14350","doi":"10.48550/arxiv.2501.14350","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.14350","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Fir- eredasr: Open-source industrial-grade mandarin speech recognition models from encoder-decoder to llm inte- gration","venue":"ArXiv.org","work_id":"62974ced-2120-413e-8766-01dae954001d","year":2025},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/2501.14350","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:7b8d291f3b27872a6eceaafb7ffaf1b89e9faa3b48ca1c22299797bcbe778455","observation_id":"81d205d0-c295-4987-b96a-312761b61059","resolution":{"observed_at":"2026-05-11T00:41:07.177531Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.12508","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T13:21:06.298615Z","title":"Fun-ASR technical report","venue":null,"work_id":"0d0144d6-599f-4fc5-a3ac-d35c144cdd41","year":2025},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:0d41e788f9c4c6860c346e48f671ac7a96ebf7f5c89977dba95a11c9a0c566dc","observation_id":"3866884e-60b9-4a87-a2c3-90a96b3ae177","resolution":{"observed_at":"2026-05-11T00:45:49.107342Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.21337","last_updated":"2026-01-30T02:58:48Z","snapshot_observed_at":"2026-07-06T22:43:26.250071Z","submitted_at":"2026-01-29T06:58:13Z","title":"Qwen3-ASR Technical Report","version":2},"cited_work":{"arxiv_id":"2601.21337","doi":"10.48550/arxiv.2601.21337","metadata_source":"pith","pith_arxiv_id":"2601.21337","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen3-ASR Technical Report","venue":"cs.CL","work_id":"db50e258-4a3d-4141-ba30-f76f8f953880","year":2026},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/2601.21337","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:9a557618e3c0efc574d56d62d9fb30539cb133a104115be346a3864c8428f93b","observation_id":"cfc7bf5b-8ff7-4428-a77f-a59e98e71120","resolution":{"observed_at":"2026-05-13T13:59:09.287362Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.02016","last_updated":"2021-10-15T18:39:21Z","snapshot_observed_at":"2026-08-06T09:07:32.396623Z","submitted_at":"2021-06-03T17:35:14Z","title":"Semantic-WER: A Unified Metric for the Evaluation of ASR Transcript for End Usability","version":2},"cited_work":{"arxiv_id":"2106.02016","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2106.02016","snapshot_observed_at":"2026-06-29T07:33:13.678073Z","title":"arXiv preprint arXiv:2106.02016 , year=","venue":null,"work_id":"c7686b6f-da3c-4434-b711-7a0fbf497551","year":2021},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/2106.02016","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:765cb667dbb1907f3915093378575e1ae5a5fbe98e44efc5cbbb6d4522a89d23","observation_id":"c459de34-28b2-4d80-8022-702a8bdd0341","resolution":{"observed_at":"2026-05-11T00:41:07.130212Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"De- noising ger: A noise-robust generative error correction with llm for speech recognition","venue":null,"work_id":"0247cd9d-bdfa-446d-89a1-95cfb12a0a7a","year":2025},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:e18c571edb53069b44a745de9676b27ee3c3c893559ad1c1e4e17c5ebaef384d","observation_id":"afd50fae-ef3e-44cb-b752-09c8455fda26","resolution":{"observed_at":"2026-05-17T04:09:00.649828Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Judging llm-as-a-judge with mt-bench and chatbot arena","venue":null,"work_id":"7cd65ced-7975-4e5b-883f-bb31fada5a32","year":2023},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:17385020981155e41b278a99c2eaf0970de8b699fa07367ca629685a0b2f8b38","observation_id":"844a8ce4-2e06-4365-805a-5a6ac0464f56","resolution":{"observed_at":"2026-05-17T04:09:00.631645Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.02138","last_updated":"2021-04-05T20:25:07Z","snapshot_observed_at":"2026-08-06T00:20:59.309289Z","submitted_at":"2021-04-05T20:25:07Z","title":"Semantic Distance: A New Metric for ASR Performance Analysis Towards Spoken Language Understanding","version":1},"cited_work":{"arxiv_id":"2104.02138","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2104.02138","snapshot_observed_at":"2026-06-29T07:33:13.676071Z","title":"Semantic distance: A new metric for asr perfor- mance analysis towards spoken language understanding","venue":null,"work_id":"90c9eab9-768d-46b4-8ed9-4c8924f2fe89","year":2021},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/2104.02138","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:34fe2dcde61952465f1e503a960f221ee4f89c966e437af202f3ece4d0f63892","observation_id":"8adf8eb3-1551-4f6d-8d01-b5fe2d78b9fb","resolution":{"observed_at":"2026-05-11T00:45:49.089046Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Laser: An llm-based asr scoring and evaluation rubric","venue":null,"work_id":"f4f76967-4148-461e-8c03-0b13a12a27e8","year":2025},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:76b91a3229dc36fc6a333bf54e13487d8e827ee1123151e84effddb9d244e138","observation_id":"a8aae583-6e9b-478c-af87-4f9f298aabc6","resolution":{"observed_at":"2026-05-17T04:09:00.636180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Multimodal error correction for speech user interfaces","venue":null,"work_id":"70d8746f-e8e4-40ec-82a1-27c97e24fc94","year":2001},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:728ba719e6c5f5483ec3d2ff4d7d9e515292c94d6ed754e544b73cde106193d1","observation_id":"c260e041-4286-4ed3-8b38-77db2e326212","resolution":{"observed_at":"2026-05-17T04:09:00.640702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Efficient speech transcription through respeaking","venue":null,"work_id":"302c7779-6bbe-4eda-83bd-c00a684de045","year":2013},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:da4c87e029893ccf991f5880023ddc6802884169c7fef580cd3c99ecbf8670f3","observation_id":"4309923e-0d77-48ee-8d5a-d1ffcc9ac3f8","resolution":{"observed_at":"2026-05-17T04:09:00.623790Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"React: Synergizing reasoning and acting in language models","venue":null,"work_id":"ae3d236c-5c5f-4357-aff8-20f27e623543","year":2022},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:b249ce783550a9a2de4a08b98e9a89b8549d015f7618b33df1898d0a902f3742","observation_id":"f45a94d5-e97c-4055-a41c-021f02f77136","resolution":{"observed_at":"2026-05-17T04:09:00.626109Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chain-of-thought prompting elicits reasoning in large language models","venue":null,"work_id":"39143a88-a0b8-4a89-8d6f-af11ee29eb04","year":2022},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:cb8b47584e8a2efd217bf5e7f9badb9a55c55025d0d5e65b14991a518411891a","observation_id":"8aae615c-0cdd-4dfa-89c0-0b24114a8cd5","resolution":{"observed_at":"2026-05-17T04:09:00.633917Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.05916","last_updated":"2020-07-12T05:38:57Z","snapshot_observed_at":"2026-07-06T09:37:29.573073Z","submitted_at":"2020-07-12T05:38:57Z","title":"The ASRU 2019 Mandarin-English Code-Switching Speech Recognition Challenge: Open Datasets, Tracks, Methods and Results","version":1},"cited_work":{"arxiv_id":"2007.05916","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.05916","snapshot_observed_at":"2026-07-02T12:46:56.918963Z","title":"The asru 2019 mandarin-english code-switching speech recognition challenge: Open datasets, tracks, methods and results","venue":null,"work_id":"fc3cad25-4fd8-45c1-b21c-8d89c78de666","year":2019},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/2007.05916","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:dc4bd77e5b8ac8fc462e6b62be61424a9452a4edda8c4088b1e9dccdd4ff845f","observation_id":"9727b7ec-bca4-4495-9d2a-728f3be473d7","resolution":{"observed_at":"2026-05-11T00:45:49.103058Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.06909","last_updated":"2021-06-13T04:09:16Z","snapshot_observed_at":"2026-07-06T11:18:44.145746Z","submitted_at":"2021-06-13T04:09:16Z","title":"GigaSpeech: An Evolving, Multi-domain ASR Corpus with 10,000 Hours of Transcribed Audio","version":1},"cited_work":{"arxiv_id":"2106.06909","doi":null,"metadata_source":"pith","pith_arxiv_id":"2106.06909","snapshot_observed_at":"2026-07-07T20:34:09.701797Z","title":"Gigaspeech: An evolving, multi-domain asr corpus with 10,000 hours of transcribed audio","venue":"cs.SD","work_id":"d786d96c-26ba-4c99-a2e9-9da6984c3cb2","year":2021},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/2106.06909","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:100974b8e256fb096fee75c6281352c149e807678bb9799d116cbc3fb76fcdd0","observation_id":"befac34a-ccf9-4788-82e2-266637a6f6f6","resolution":{"observed_at":"2026-05-11T00:41:07.146906Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Wenetspeech: A 10000+ hours multi-domain mandarin corpus for speech recognition","venue":null,"work_id":"1a25bd5a-683b-486e-a67e-44102a2ba687","year":2022},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:26da7d7b58818e943cf1f8940b7a6be02955d01d5b92451273f5ef80d90be627","observation_id":"d50269c4-9d56-4b1d-94e8-8ae91289bf92","resolution":{"observed_at":"2026-05-17T04:09:00.616351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:cb682f4057d18ad5324eb4c69485b83f56b687b5436c4fe0575a1a19cf731451","observation_id":"10534bd1-1b35-4090-8192-5db4ae352617","resolution":{"observed_at":"2026-05-11T00:41:07.138391Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.05512","last_updated":"2025-02-08T10:23:20Z","snapshot_observed_at":"2026-08-05T17:33:03.736753Z","submitted_at":"2025-02-08T10:23:20Z","title":"IndexTTS: An Industrial-Level Controllable and Efficient Zero-Shot Text-To-Speech System","version":1},"cited_work":{"arxiv_id":"2502.05512","doi":"10.48550/arxiv.2502.05512","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.05512","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"IndexTTS: An Industrial-Level Controllable and Efficient Zero-Shot Text-To-Speech System","venue":"ArXiv.org","work_id":"94f6526d-bd45-4dd3-b4b1-e8f9bb61768a","year":2025},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"cited_paper":"/paper/2502.05512","citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:369073c364913f1dc8ebb4016ce0f233ed2634e529906317d62639764df2a5e2","observation_id":"066c6c75-1c36-4075-ab14-f14e9810b178","resolution":{"observed_at":"2026-05-11T00:41:07.162050Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Zechner and K","venue":null,"work_id":"88c21e4b-1740-4a03-a9b9-452ea68c62fd","year":2019},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:41366e36d7c2280e541dfb573e834297b34c3bd20bbf25493b26a9f522b2cb3c","observation_id":"99a25ce7-2aec-46f0-b7be-8f1c021479e5","resolution":{"observed_at":"2026-05-17T04:09:00.638340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"1895.0041","doi":"10.1098/rspl.1895.0041","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Proceedings of the Royal Society of London 58, 240–242","venue":"Proceedings of the Royal Society of London","work_id":"4f6bd244-2fca-4572-a60c-38ca20802cf2","year":null},"citing_paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-10T18:22:08.670559Z"},"links":{"citing_paper":"/paper/2604.09121"},"observation_digest":"sha256:510b64f49868708533e20c7e0c1f9d16198fead1127653d613e21977b0a2d267","observation_id":"a2ab57b1-b9d6-4151-8b8f-f8969f6d003c","resolution":{"observed_at":"2026-05-10T18:25:42.967159Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.09121","last_updated":"2026-04-14T06:45:50Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-02T20:05:01.635841Z","submitted_at":"2026-04-10T09:02:42Z","title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":1,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":2,"verified_exact":13,"verified_fuzzy":19},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 2 inbound Pith citation observations for arXiv:2604.09121."}