{"as_of":"2026-08-19T23:12:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:fced468955073e7a36cf96b27227798d6a1d9fdcf5d1eec07ac3d0a817856892","coverage":[{"denominator":37,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T16:21:50.792032Z","state":"measured"},{"denominator":50,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":50,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":13,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":13,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T10:33:10.833843Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T11:49:50.788200Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-08-16T10:33:10.833843Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.18039","last_updated":"2025-09-14T07:36:28Z","snapshot_observed_at":"2026-08-18T05:05:41.512855Z","submitted_at":"2025-04-25T03:12:43Z","title":"MultiMind: Enhancing Werewolf Agents with Multimodal Reasoning and Theory of Mind","version":4},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-16T10:33:10.833843Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2504.18039"},"observation_digest":"sha256:93c055c056c0c4c0061257f5de384b3e53cc2f8407735e9d4d8f597923c3db8d","observation_id":"06cb53d3-6404-4a88-af11-6ebc594f1ca1","resolution":{"observed_at":"2026-08-16T10:33:10.833843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":"2501.13306","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-07-04T11:49:50.788200Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","venue":null,"work_id":"45f98848-4dc2-4c34-9aac-786ca4cd4ace","year":2025},"citing_paper":{"arxiv_id":"2504.18425","last_updated":"2025-04-25T15:31:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-25T15:31:46Z","title":"Kimi-Audio Technical Report","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-11T19:21:26.933349Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2504.18425"},"observation_digest":"sha256:f70a0c815496d9ff1d6785bf7ec32180faa6f999f51a227a8fa0ff4d4eb9c3ea","observation_id":"203b544f-3836-4423-a139-4e5a5f95024d","resolution":{"observed_at":"2026-05-11T19:21:27.283015Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-08-07T10:18:51.939607Z","title":"Osum: Advancing open speech un- derstanding models with limited resources in academia,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05796","last_updated":"2025-06-06T06:43:34Z","snapshot_observed_at":"2026-08-15T22:09:59.303146Z","submitted_at":"2025-06-06T06:43:34Z","title":"Diarization-Aware Multi-Speaker Automatic Speech Recognition via Large Language Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:51.939607Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2506.05796"},"observation_digest":"sha256:992daf777e49c8c7337a4824601f6b0f5874fe0b58b9273d5d9d0981877b009a","observation_id":"314f9479-bc79-4f09-bcb5-5fa240649b68","resolution":{"observed_at":"2026-08-07T10:18:51.939607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-08-05T20:59:46.041859Z","title":"What do you think I should eat?","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.09599","last_updated":"2026-05-23T09:18:10Z","snapshot_observed_at":"2026-08-19T16:32:44.391622Z","submitted_at":"2025-08-13T08:28:21Z","title":"BridgeTA: Bridging the Representation Gap in Knowledge Distillation via Teacher Assistant for Bird's Eye View Map Segmentation","version":2},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-05T20:59:46.041859Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2508.09599"},"observation_digest":"sha256:014496b3aec2d774626aaf6ef47b8c30f2488d15ff52ee6bdbf8da838e1b7c34","observation_id":"27882c48-9fd5-4dce-80ee-c7519ae63114","resolution":{"observed_at":"2026-08-05T20:59:46.041859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-08-05T21:03:09.640862Z","title":"What do you think I should eat?","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.09600","last_updated":"2025-09-03T13:33:34Z","snapshot_observed_at":"2026-08-19T07:13:22.852048Z","submitted_at":"2025-08-13T08:30:14Z","title":"OSUM-EChat: Enhancing End-to-End Empathetic Spoken Chatbot via Understanding-Driven Spoken Dialogue","version":2},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-05T21:03:09.640862Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2508.09600"},"observation_digest":"sha256:d99d671d20b30db2bbc8d647020cefc77deda64b3f555f14285506e45fc00cbd","observation_id":"787c8920-5478-449b-bb67-7791784322b4","resolution":{"observed_at":"2026-08-05T21:03:09.640862Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":"2501.13306","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-07-04T11:49:50.788200Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","venue":null,"work_id":"45f98848-4dc2-4c34-9aac-786ca4cd4ace","year":2025},"citing_paper":{"arxiv_id":"2509.14804","last_updated":"2025-09-18T09:59:55Z","snapshot_observed_at":"2026-08-07T14:31:44.422604Z","submitted_at":"2025-09-18T09:59:55Z","title":"Towards Building Speech Large Language Models for Multitask Understanding in Low-Resource Languages","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-18T16:27:37.596817Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2509.14804"},"observation_digest":"sha256:cd7fd7ff2ff56b07a4aec67ce8dd896c2a80b1bac6dafce4a939d041e71bafbf","observation_id":"93284ac9-394a-42dc-9ced-1d4b75bc47dd","resolution":{"observed_at":"2026-05-18T16:31:37.228057Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-08-03T17:16:38.479016Z","title":"Osum: Advancing open speech understanding models with limited resources in academia","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.10324","last_updated":"2026-06-22T06:41:41Z","snapshot_observed_at":"2026-08-17T16:07:27.921903Z","submitted_at":"2025-12-11T06:18:58Z","title":"EchoingPixels: Aliasing-Resistant Joint Token Reduction for Audio-Visual LLMs","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-03T17:16:38.479016Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2512.10324"},"observation_digest":"sha256:b74405f63ef322d7f7a56c1aef0f0be3befbbbbbb7aaedf90b9c04c37d675484","observation_id":"267ba0df-de4f-406e-956d-5b4178a6c813","resolution":{"observed_at":"2026-08-03T17:16:38.479016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":"2501.13306","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-07-04T11:49:50.788200Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","venue":null,"work_id":"45f98848-4dc2-4c34-9aac-786ca4cd4ace","year":2025},"citing_paper":{"arxiv_id":"2604.11594","last_updated":"2026-04-24T06:36:24Z","snapshot_observed_at":"2026-08-14T21:13:08.485449Z","submitted_at":"2026-04-13T15:06:05Z","title":"HumDial-EIBench: A Human-Recorded Multi-Turn Emotional Intelligence Benchmark for Audio Language Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T15:24:27.118694Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2604.11594"},"observation_digest":"sha256:285b9252377e2e1a6a6bc49257b69b996b07772b792a5226d9bae9803cc2bfcb","observation_id":"196a7acf-ff2b-4f85-a4b7-7f2990d536e4","resolution":{"observed_at":"2026-05-11T10:36:05.898403Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":"2501.13306","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-07-04T11:49:50.788200Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","venue":null,"work_id":"45f98848-4dc2-4c34-9aac-786ca4cd4ace","year":2025},"citing_paper":{"arxiv_id":"2604.12527","last_updated":"2026-07-03T08:49:19Z","snapshot_observed_at":"2026-07-12T21:18:44.451500Z","submitted_at":"2026-04-14T10:00:39Z","title":"Audio-Cogito: Towards Deep Audio Reasoning in Large Audio Language Models","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T14:10:03.707886Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2604.12527"},"observation_digest":"sha256:7eba07ec9ab877fe9ad818f06223618c3ba70c9a02b36803b26618eec9b492ee","observation_id":"6ead0d2e-a824-4f9f-951d-49a72e4765b3","resolution":{"observed_at":"2026-05-10T14:10:28.342560Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-07-12T21:18:46.566338Z","title":"Osum: Advancing open speech understand- ing models with limited resources in academia,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.12527","last_updated":"2026-07-03T08:49:19Z","snapshot_observed_at":"2026-07-12T21:18:44.451500Z","submitted_at":"2026-04-14T10:00:39Z","title":"Audio-Cogito: Towards Deep Audio Reasoning in Large Audio Language Models","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-12T21:18:46.566338Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2604.12527"},"observation_digest":"sha256:1a2ebb96cc806ad2e00da7590a23dacd2b5ddde9b8d5198d0208ddd6f85c4fd9","observation_id":"d2db76cc-80e9-4dfd-93d8-fbf512f19a12","resolution":{"observed_at":"2026-07-12T21:18:46.566338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":"2501.13306","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-07-04T11:49:50.788200Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","venue":null,"work_id":"45f98848-4dc2-4c34-9aac-786ca4cd4ace","year":2025},"citing_paper":{"arxiv_id":"2604.18204","last_updated":"2026-04-20T12:54:14Z","snapshot_observed_at":"2026-07-06T23:05:08.728446Z","submitted_at":"2026-04-20T12:54:14Z","title":"Hard to Be Heard: Phoneme-Level ASR Analysis of Phonologically Complex, Low-Resource Endangered Languages","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T04:25:37.725216Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2604.18204"},"observation_digest":"sha256:a8f56c7a98f4af01c64c8aff5baee76b52dcda375d3e816b19ff589efc147776","observation_id":"ae99d2f1-9b81-426c-9ce1-96113b53d4bc","resolution":{"observed_at":"2026-05-11T11:56:31.006517Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":"2501.13306","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-07-04T11:49:50.788200Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","venue":null,"work_id":"45f98848-4dc2-4c34-9aac-786ca4cd4ace","year":2025},"citing_paper":{"arxiv_id":"2606.22868","last_updated":"2026-06-22T05:24:35Z","snapshot_observed_at":"2026-08-14T11:36:02.442036Z","submitted_at":"2026-06-22T05:24:35Z","title":"MSU-Bench: Towards Speaker-Centric Understanding in Conversational Multi-Speaker Scenarios","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-26T07:36:34.307652Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2606.22868"},"observation_digest":"sha256:4628f7314a2d0be8a3468928e95da895fd4e95320ec514a4660d8dfa7f122d2e","observation_id":"760bc7e5-e472-4085-b43d-a2c487a220c8","resolution":{"observed_at":"2026-07-04T11:49:50.790654Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-08-01T05:50:29.618737Z","title":"OSUM: Advancing open speech understand- ing models with limited resources in academia,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.22100","last_updated":"2026-07-24T08:52:44Z","snapshot_observed_at":"2026-08-13T04:53:52.000451Z","submitted_at":"2026-07-24T08:52:44Z","title":"MEUSLI: a Multilingual Projector for LLM-based ASR and Beyond","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-01T05:50:29.618737Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2607.22100"},"observation_digest":"sha256:4c51eef806c1002bfd1d7c0c60c5809711c2d665b83d6641d914454400641836","observation_id":"459daaf3-3a19-4091-a8cc-2608f1eb74b1","resolution":{"observed_at":"2026-08-01T05:50:29.618737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.13306/citation-record","integrity":"/paper/2501.13306/integrity","json":"/paper/2501.13306/citation-record.json","paper":"/paper/2501.13306"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.249877Z","title":"Data Products , 2024","venue":null,"work_id":"7e4b687f-bc37-41a9-b5b0-284609ce6c0b","year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.641410Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:157da0c63f509cab2bef5ac8b73c99c4a20abdc573315d0c8fa150a16e20c103","observation_id":"c4845ecc-ab65-4973-bb13-35ee447b6f80","resolution":{"observed_at":"2026-08-10T16:21:51.254183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04051","last_updated":"2024-07-11T02:08:35Z","snapshot_observed_at":"2026-08-16T13:36:59.551255Z","submitted_at":"2024-07-04T16:49:02Z","title":"FunAudioLLM: Voice Understanding and Generation Foundation Models for Natural Interaction Between Humans and LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04051","snapshot_observed_at":"2026-08-10T16:21:50.645939Z","title":"FunaudioLLM : Voice understanding and generation foundation models for natural interaction between humans and LLMs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.645939Z"},"links":{"cited_paper":"/paper/2407.04051","citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:abace3e9d2a9165513c3cbff9bf5b44eb47f574d7c48cf3a38c4576715852dd3","observation_id":"0a0579ea-5f4d-4fa8-a1e7-9c9f4ee76847","resolution":{"observed_at":"2026-08-10T16:21:50.645939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-08-09T21:25:20.369782Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-10T16:21:50.650773Z","title":"Qwen technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.650773Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:1ebb027d5fba4b1a1c034b02ddac218aeba158b75d2fdef49d629a501b656f44","observation_id":"26405e1f-01c6-4177-9c8e-34b8d827b0a4","resolution":{"observed_at":"2026-08-10T16:21:50.650773Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.237562Z","title":"AISHELL-1 : An open-source mandarin speech corpus and a speech recognition baseline","venue":null,"work_id":"0f888a07-4cef-42ff-9682-1aa79db5509b","year":2017},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.655249Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:9930c7dcbdca1714428c2fd5e96292b29ecbfeaf4712ac0a77f284dceab46805","observation_id":"1d2f7d2d-bf49-4319-9d59-0b74ce33614c","resolution":{"observed_at":"2026-08-10T16:21:51.241788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.225589Z","title":"IEMOCAP : Interactive emotional dyadic motion capture database","venue":null,"work_id":"7ae0075c-6600-426d-a73d-7af303b115fb","year":2008},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.660136Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:abb91d8667056f8f88635a08e646bff8cda60f5556ccec940253c29a419698fe","observation_id":"46728167-1fc3-45b8-b624-2198cb239ab6","resolution":{"observed_at":"2026-08-10T16:21:51.229299Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.213771Z","title":"MSP-IMPROV : An acted corpus of dyadic interactions to study emotion perception","venue":null,"work_id":"f7d06017-e4ae-42dd-97f1-f5df75210cd6","year":2017},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.664124Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:1688aea4e73f56ddec513357a73f02356ca64d3bbfef576688cea146a846a9fb","observation_id":"52440562-690e-4e16-84cb-b880bc7a35ee","resolution":{"observed_at":"2026-08-10T16:21:51.217573Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07919","last_updated":"2023-12-21T10:20:42Z","snapshot_observed_at":"2026-08-07T10:17:55.688598Z","submitted_at":"2023-11-14T05:34:50Z","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07919","snapshot_observed_at":"2026-08-10T16:21:50.668642Z","title":"Qwen-Audio : Advancing universal audio understanding via unified large-scale audio-language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.668642Z"},"links":{"cited_paper":"/paper/2311.07919","citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:06e71e9a94b8b88d535385a6e0308f071412f826b080e9bd58695c7f799c8506","observation_id":"15110fa2-db69-4e3d-a69b-0a745922a32d","resolution":{"observed_at":"2026-08-10T16:21:50.668642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10759","last_updated":"2024-07-15T14:38:09Z","snapshot_observed_at":"2026-08-14T01:27:16.843576Z","submitted_at":"2024-07-15T14:38:09Z","title":"Qwen2-Audio Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10759","snapshot_observed_at":"2026-08-10T16:21:50.672430Z","title":"Qwen2-Audio technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.672430Z"},"links":{"cited_paper":"/paper/2407.10759","citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:c644d11ed5987b063e7dddf79b56d8044b57ead14488fcac804edc9b8c211718","observation_id":"953d4eb8-70f3-4527-b300-24d2dec5e61f","resolution":{"observed_at":"2026-08-10T16:21:50.672430Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.202245Z","title":"Data products, 2024","venue":null,"work_id":"17423639-be07-4a44-9542-5af13e4d23a0","year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.676243Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:38a5aeaac6733d806397d5d835df81d3bb525275a6b651c399268c6e7d85edb5","observation_id":"2b3e2324-61fc-46a6-8ea9-9c3b754f2a8b","resolution":{"observed_at":"2026-08-10T16:21:51.205841Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.191589Z","title":"Data products, 2024","venue":null,"work_id":"403df258-27d9-4462-90b3-1e0b86e937b0","year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.679557Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:8ffd72d05cac309df162e3975a6cfc9d21d6a03301ec2713e87d19fa2e7ffc6e","observation_id":"0013ddfe-9b79-4433-88e9-fedde57b0f45","resolution":{"observed_at":"2026-08-10T16:21:51.194923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1808.10583","last_updated":"2018-09-13T02:45:27Z","snapshot_observed_at":"2026-08-14T18:34:46.662561Z","submitted_at":"2018-08-31T03:11:08Z","title":"AISHELL-2: Transforming Mandarin ASR Research Into Industrial Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1808.10583","snapshot_observed_at":"2026-08-10T16:21:50.682871Z","title":"AISHELL-2 : Transforming Mandarin ASR research into industrial scale","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.682871Z"},"links":{"cited_paper":"/paper/1808.10583","citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:edaa7bea727e11bdb58b8b52f715dc6e55e0fea281200b97d5af10b844b812fe","observation_id":"ef476ca6-1f6f-4dad-8424-4384230db56e","resolution":{"observed_at":"2026-08-10T16:21:50.682871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.180883Z","title":"Gemmeke, Daniel P","venue":null,"work_id":"bf5f96ab-ab6d-4e85-8628-4873ae7dbf2c","year":2017},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.688136Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:6340b7f4d06956d4ee7cd43e3d330db1412db71bbfd9fe5cdcc21f7b6c19f80a","observation_id":"02eea089-deb6-4bff-ab24-2e277fa8f5e9","resolution":{"observed_at":"2026-08-10T16:21:51.184074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.170190Z","title":"Vocalsound: A dataset for improving human vocal sounds recognition","venue":null,"work_id":"de60ac5e-9cae-471a-b296-9c60af4f317a","year":2022},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.692800Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:4dfe7fc29fcae7f57fb2bb4d8c925786687a13454a626a991d509e07054f0d9f","observation_id":"b763e6cd-8fb7-45a2-ac99-ef753df47ed6","resolution":{"observed_at":"2026-08-10T16:21:51.173901Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.160109Z","title":"LoRA : Low -rank adaptation of large language models","venue":null,"work_id":"f4015ca5-ef89-4fc8-a04d-8dcc538500e5","year":2022},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.697178Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:660987c81dd4bfa98b5c685c70c4edcad0806924baee8ae7bd2f91cff7afed98","observation_id":"f7be6a17-c051-4edf-a4c8-4012dc488c1a","resolution":{"observed_at":"2026-08-10T16:21:51.163633Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.149627Z","title":"Datasets, 2017","venue":null,"work_id":"2b9f3c6a-3283-41fb-8744-26f4fbeac048","year":2017},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.701458Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:f8fc2511bbb3b3ddd49faebfaab1c3e1378970ae35b1dca76b263204cbd509d5","observation_id":"d2030894-3781-4c39-91b1-699135cd2541","resolution":{"observed_at":"2026-08-10T16:21:51.153105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.138763Z","title":"Schuller, and Jianhua Tao","venue":null,"work_id":"e74e919d-66f2-46c5-87ce-ed492b9ddb1c","year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.705820Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:524b5906414bf258088d0cac946a3e82aace2ffe0f260ff283e1817ec71996db","observation_id":"b228b375-8c03-4806-9680-93ae9f01d650","resolution":{"observed_at":"2026-08-10T16:21:51.142135Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.127916Z","title":"Emotion2vec: Self-supervised pre-training for speech emotion representation","venue":null,"work_id":"fffd6da1-f113-4fd6-80c8-2c6897ebd418","year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.709957Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:c2886d03f782b8016b2a3791236fa33aca080c53023b2b8acf53e878152f1261","observation_id":"f66d6d84-5633-41ca-9be7-69b3c0021db3","resolution":{"observed_at":"2026-08-10T16:21:51.131662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.116237Z","title":"The MSP -conversation corpus","venue":null,"work_id":"fc6677a1-c986-40ce-be9e-e1170b92664f","year":2020},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.714460Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:caf0bf59636bb51298d76e2582a546ea3216c7b7e387fc77f04da998d5ba253e","observation_id":"fdfd17ed-1cf7-4c0e-be38-c59aaeadb688","resolution":{"observed_at":"2026-08-10T16:21:51.120340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.104166Z","title":"MAGICDATA mandarin Chinese read speech corpus, 2019","venue":null,"work_id":"16a846a7-921e-4a74-b02c-43bce049a01f","year":2019},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.718846Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:c92b2b2cff2511e86dcbda38c5e2dacf31f1931d96345add8a5786bf8804e08b","observation_id":"f9167552-5a2f-4ed4-9def-0c061fd591f3","resolution":{"observed_at":"2026-08-10T16:21:51.107988Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.092321Z","title":"Librispeech: An ASR corpus based on public domain audio books","venue":null,"work_id":"d6ce973a-122a-4766-863c-5feb5f37bdb9","year":2015},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.724151Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:4847c3b8fd4dbd2a6c79ffb9a8ed2b4f03f7043707aba4a77bc8f3afac90585e","observation_id":"a9ba5083-904c-4640-9995-ae09ad4aef9e","resolution":{"observed_at":"2026-08-10T16:21:51.096486Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.079562Z","title":"Reproducing whisper-style training using an open-source toolkit and publicly available data","venue":null,"work_id":"4d96c3b4-63fa-4c2a-83f5-94d64077a80c","year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.728388Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:48b88c0263f26b929ae1eefebf3c54c349cb5778e8fa8cd43bbc8aaf5ff3f023","observation_id":"40c2663a-8e81-4e98-b77a-ac501ac14610","resolution":{"observed_at":"2026-08-10T16:21:51.084881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.069405Z","title":null,"venue":null,"work_id":"d0745753-371a-439a-bf2b-ed742826e9d1","year":2015},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.731940Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:42e36940811c06e47767d145fd342fe62537d5bec5d6783b1c72cf76920e823f","observation_id":"a5df2899-512f-413b-933e-4b6cfa6d6f51","resolution":{"observed_at":"2026-08-10T16:21:51.072413Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.058256Z","title":"MELD : A multimodal multi-party dataset for emotion recognition in conversations","venue":null,"work_id":"e9eb1177-7fd9-4e93-9d20-448dd595548e","year":2019},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.735946Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:0e396a92a3205e97fcd0208efc2c02ca7ea6186d2f1786e4656f96d12fdc3af1","observation_id":"0c1bfbc9-4cb3-435b-9387-85f3af8b9463","resolution":{"observed_at":"2026-08-10T16:21:51.061704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.047668Z","title":"Robust speech recognition via large-scale weak supervision","venue":null,"work_id":"72976760-3f8c-44c6-becb-20432d427cc2","year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.739872Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:6fde23f78a81f7fa05f6fcb6532d4ddb4bf2833e07ca5bafe500c956f41678e9","observation_id":"c4678dbb-125c-49d6-8e58-cd612f3a968a","resolution":{"observed_at":"2026-08-10T16:21:51.051351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.035672Z","title":"Nonspeech7k dataset: Classification and analysis of human non-speech sound","venue":null,"work_id":"5de055d4-fbd8-4e61-a7a7-b0facfcf9ffb","year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.743915Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:a77ab091416d1df0abda385f488e2a15264adeb08b47c6aa7e5a5314eda7934e","observation_id":"495dd3a5-d28c-4dd6-a54d-f0e27ab3d062","resolution":{"observed_at":"2026-08-10T16:21:51.039921Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.05916","last_updated":"2020-07-12T05:38:57Z","snapshot_observed_at":"2026-08-16T15:37:12.257609Z","submitted_at":"2020-07-12T05:38:57Z","title":"The ASRU 2019 Mandarin-English Code-Switching Speech Recognition Challenge: Open Datasets, Tracks, Methods and Results","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.05916","snapshot_observed_at":"2026-08-10T16:21:50.748041Z","title":"The ASRU 2019 Mandarin - English code-switching speech recognition challenge: Open datasets, tracks, methods and results","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.748041Z"},"links":{"cited_paper":"/paper/2007.05916","citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:2c267fa5249beeb1ee302401ff72287928fffd7f1b1dda2d33eab681d9ea1740","observation_id":"64e3ad61-25f5-4207-90f2-ffa3e92de62b","resolution":{"observed_at":"2026-08-10T16:21:50.748041Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.022090Z","title":"Achieving timestamp prediction while recognizing with non-autoregressive end-to-end ASR model","venue":null,"work_id":"40694906-be51-42c2-a10c-380684264462","year":2022},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.751915Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:b38ff11af951e4ad365d276f10b413f872f33e5eedf82241cff13a388b2c3643","observation_id":"e20f0866-7a74-4ce1-8c57-5d36fc155563","resolution":{"observed_at":"2026-08-10T16:21:51.026269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15622","last_updated":"2024-12-20T07:28:04Z","snapshot_observed_at":"2026-08-13T18:08:12.828161Z","submitted_at":"2024-12-20T07:28:04Z","title":"TouchASP: Elastic Automatic Speech Perception that Everyone Can Touch","version":1},"cited_work":{"arxiv_id":"2412.15622","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.15622","snapshot_observed_at":"2026-08-10T16:21:50.826977Z","title":"TouchASP: Elastic Automatic Speech Perception that Everyone Can Touch","venue":"eess.AS","work_id":"87dfe593-0db0-4e11-aca7-24a3479f58d4","year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.755899Z"},"links":{"cited_paper":"/paper/2412.15622","citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:60cdf2c6038e44a6eec4660fe443d8f6270e00c91ad32b065ef7606b5c2f3a61","observation_id":"afe57756-a2d3-470d-80f4-fd23a3dffd5b","resolution":{"observed_at":"2026-08-10T16:21:50.833353Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.010976Z","title":"PandaGPT : One model to instruction-follow them all","venue":null,"work_id":"96bcb1d9-e852-440b-a873-ea42aaa6f101","year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.760638Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:191ff9c6da89084100e27defbe499213ed1672285681bbef67e6fedc2632d045","observation_id":"84ec1fc9-4e59-48ac-8f98-fc5b362f883e","resolution":{"observed_at":"2026-08-10T16:21:51.014485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.999115Z","title":"SALMONN : Towards generic hearing abilities for large language models","venue":null,"work_id":"f1be6610-7358-4cc1-8412-69ea2b933e4d","year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.764260Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:908be733a4d2f4719f8d4e786dba89d1923a52411cad395b7bccfff42dbb7577","observation_id":"5fcabe55-126a-48c5-bc0c-739cc90a3f2d","resolution":{"observed_at":"2026-08-10T16:21:51.002925Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.987891Z","title":"Kespeech: An open source speech dataset of Mandarin and its eight subdialects","venue":null,"work_id":"a4b5bd2b-8121-4472-a295-252c46c18dc7","year":2021},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.767988Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:f44d4bb1113e380d7c83332940b327bcf438e152f4282fa15b45e355134caf12","observation_id":"2f9dd831-0c14-453c-9f29-ec7c9ad4fe32","resolution":{"observed_at":"2026-08-10T16:21:50.992013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.976037Z","title":"Upadhyay, Woan-Shiuan Chien, Bo-Hao Su, Lucas Goncalves, Ya-Tse Wu, Ali N","venue":null,"work_id":"7cd8f3e0-f6f0-41cf-b2d2-2855ad84d677","year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.771903Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:f2184a30423cd6d13daa5f8dcf994f72d8044709a1d88d5977ae8fd8a24ff6bf","observation_id":"5eef7f39-a106-4968-9c33-982a1ac3b2c2","resolution":{"observed_at":"2026-08-10T16:21:50.980216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.963922Z","title":"Attention is all you need","venue":null,"work_id":"56d35d8f-a3e5-4b4a-85cc-decadcc9a0d3","year":2017},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.775557Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:beda6bb1ee1998552386662019f8d191943750d6a07c4353b399d3bd6c974b45","observation_id":"52d13f61-9502-479a-936d-b5b890a9a7ae","resolution":{"observed_at":"2026-08-10T16:21:50.967826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.951780Z","title":"A large-scale Chinese short-text conversation dataset","venue":null,"work_id":"39163cb5-0043-49ba-8504-0f27c5a40a8c","year":2020},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.779084Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:36659315dc871efac30f3c4efce52739aae26f78aa8089c23410d5f27bd64e97","observation_id":"caff80b2-1551-4bc4-b5e7-265f7e582f6b","resolution":{"observed_at":"2026-08-10T16:21:50.955944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.939016Z","title":"WENETSPEECH : A 10000+ hours multi-domain Mandarin corpus for speech recognition","venue":null,"work_id":"aaa35b52-4127-45a4-9137-182805fe20ae","year":2022},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.783589Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:f34dca72381f3d5bba4437a26282eee7b858f51642b2c2bbc035b45a2b3776ec","observation_id":"75834008-8b63-4873-963e-aa9b98735c35","resolution":{"observed_at":"2026-08-10T16:21:50.943291Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.926518Z","title":"M 3 ED : Multi -modal multi-scene multi-label emotional dialogue database","venue":null,"work_id":"fd28aebe-16d7-4b50-bd12-7d797cc66aa4","year":2022},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.788155Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:0d63af93a0897fe712007cef42d4aa6c77bea2daae47b0ac68275e3676b60540","observation_id":"98512b3b-063f-4236-be2d-d25226d42cf0","resolution":{"observed_at":"2026-08-10T16:21:50.930655Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.912713Z","title":"Seen and unseen emotional style transfer for voice conversion with a new emotional speech dataset","venue":null,"work_id":"a59510a2-5786-40c4-b5f6-cae484b15cea","year":2021},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.792032Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:278a5d01d1f0c8a71ef3c0b5d12cf75a443e6d8f336972b3782e2c301af49070","observation_id":"26a6b988-ecf9-42c2-8381-48f36e2c8c64","resolution":{"observed_at":"2026-08-10T16:21:50.916362Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","latest_version":2,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-15T06:20:21.428418Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia"},"reference_resolution":{"displayed":37,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":1,"verified_fuzzy":29},"total_outbound_references":37},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 37 of 37 outbound references and 13 inbound Pith citation observations for arXiv:2501.13306."}