{"as_of":"2026-08-06T19:29:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:955670ba38fb70ef7ae18081f493804f57bba8ee55b52523e280ed20bbbfe8c9","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":40,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":40,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":40,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":40,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T19:02:41.013510Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2507.23511","last_updated":"2026-05-11T14:54:52Z","snapshot_observed_at":"2026-08-03T03:37:10.882228Z","submitted_at":"2025-07-31T12:47:43Z","title":"MECAT: A Multi-Experts Constructed Benchmark for Fine-Grained Audio Understanding Tasks","version":3},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-19T02:41:52.996457Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2507.23511"},"observation_digest":"sha256:8156e1196a41d4c968aacb7817047ab420f16925d18969adbc6e47969bcfab96","observation_id":"c0ea1ff9-b90e-4dea-b21d-9771092d2392","resolution":{"observed_at":"2026-05-19T02:41:59.617089Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2510.00626","last_updated":"2026-04-26T15:50:39Z","snapshot_observed_at":"2026-07-06T22:31:20.110402Z","submitted_at":"2025-10-01T07:59:45Z","title":"When Silence Matters: The Impact of Irrelevant Audio on Text Reasoning in Large Audio-Language Models","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-18T11:08:08.916893Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2510.00626"},"observation_digest":"sha256:e70ca920c200905ad571040854c1ca6beaf2c88aa31a726f3da1ee921169b828","observation_id":"152eed71-91e6-42da-9dec-b4226d8ff683","resolution":{"observed_at":"2026-05-18T11:11:17.876150Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-04T11:30:24.529386Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.04584","last_updated":"2026-06-24T15:08:01Z","snapshot_observed_at":"2026-08-04T11:30:22.923124Z","submitted_at":"2025-10-06T08:36:17Z","title":"Robustness assessment of large audio language models in multiple-choice evaluation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T11:30:24.529386Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2510.04584"},"observation_digest":"sha256:9f3820996011f31a7ced7a456e70515898083b94d6a21ca4765353ad625b07b5","observation_id":"aa23bad6-c2cb-4114-9261-b27d6cec8f14","resolution":{"observed_at":"2026-08-04T11:30:24.529386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-04T11:23:19.304868Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.05478","last_updated":"2026-06-08T05:16:04Z","snapshot_observed_at":"2026-08-04T21:47:56.469923Z","submitted_at":"2025-10-07T00:39:14Z","title":"AQA-TTRL: Self-Adaptation in Audio Question Answering with Test-Time Reinforcement Learning","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T11:23:19.304868Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2510.05478"},"observation_digest":"sha256:6ebb9b34fb895b4c10735f27549019691d43533c0bf9a092fd22e41168ea3466","observation_id":"f721397d-e7b9-4569-98c5-5247d43af320","resolution":{"observed_at":"2026-08-04T11:23:19.304868Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-04T10:13:42.663241Z","title":"Nasrin Mostafazadeh, Nathanael Chambers, Xiaodong He, Devi Parikh, Dhruv Batra, Lucy Vanderwende, Pushmeet Kohli, and James Allen","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.11098","last_updated":"2026-07-06T08:06:32Z","snapshot_observed_at":"2026-08-04T21:47:58.063486Z","submitted_at":"2025-10-13T07:45:52Z","title":"VCB Bench: An Evaluation Benchmark for Audio-Grounded Large Language Model Conversational Agents","version":5},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T10:13:42.663241Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2510.11098"},"observation_digest":"sha256:38ec2c1c4d18ac25f7c3f5ab55360efabf86126ed8e06eafada86a279cfb8cf0","observation_id":"79175ecb-6a78-446b-aaad-b157ba509d01","resolution":{"observed_at":"2026-08-04T10:13:42.663241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-03T21:06:10.705003Z","title":"Mmar: A challenging benchmark for deep rea- soning in speech, audio, music, and their mix.arXiv preprint arXiv:2505.13032,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.16757","last_updated":"2026-06-29T23:59:45Z","snapshot_observed_at":"2026-08-03T21:06:06.357839Z","submitted_at":"2025-11-20T19:17:35Z","title":"Revisiting Audio-language Pretraining for Learning General-purpose Audio Representation","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-03T21:06:10.705003Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2511.16757"},"observation_digest":"sha256:4df38d3d03084b8ad5fb25b3c44a3d3d5d26fa3744d7f98bb11eb16378463bd4","observation_id":"c171446b-c51c-4493-9a76-50842f872cfe","resolution":{"observed_at":"2026-08-03T21:06:10.705003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-03T19:38:24.214906Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.09066","last_updated":"2026-06-29T11:40:23Z","snapshot_observed_at":"2026-08-06T03:12:02.087196Z","submitted_at":"2025-11-28T14:41:48Z","title":"ORCA: Open-ended Response Correctness Assessment for Audio Question Answering","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-03T19:38:24.214906Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2512.09066"},"observation_digest":"sha256:ca9555e8ea6e842d2b8124b4af42a803dc0c6524bd841935f0d0e25fc8a89e2e","observation_id":"0641492d-ef98-48b5-a9d1-1df261d82689","resolution":{"observed_at":"2026-08-03T19:38:24.214906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2601.02731","last_updated":"2026-04-29T02:44:11Z","snapshot_observed_at":"2026-07-06T22:40:46.465095Z","submitted_at":"2026-01-06T05:49:41Z","title":"Omni2Sound: Towards Unified Video-Text-to-Audio Generation","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-16T17:25:38.591071Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2601.02731"},"observation_digest":"sha256:e4951e8312c88a5132b8c5e4ee2cb68e43143bc8fece18f80bdd1d72628e65fb","observation_id":"474c89ad-ff40-4a63-afa2-9db29c20e7f5","resolution":{"observed_at":"2026-05-16T17:28:10.049255Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2601.12248","last_updated":"2026-05-10T16:47:20Z","snapshot_observed_at":"2026-07-06T22:42:03.665045Z","submitted_at":"2026-01-18T03:55:28Z","title":"AQUA-Bench: Beyond Finding Answers to Knowing When There Are None in Audio Question Answering","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-16T14:04:17.935630Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2601.12248"},"observation_digest":"sha256:fe80d48632f7c13aa744f1fbddf42cba5b712f4d028ca22c6a3375f1e734347c","observation_id":"6ac55984-e2be-4519-9b7a-c4ef584c94d0","resolution":{"observed_at":"2026-05-16T14:07:58.685833Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2602.07064","last_updated":"2026-04-07T13:49:35Z","snapshot_observed_at":"2026-08-03T13:22:02.057620Z","submitted_at":"2026-02-05T14:04:51Z","title":"OmniFysics: Towards Physical Intelligence Evolution via Omni-Modal Signal Processing and Network Optimization","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-16T07:09:46.254851Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2602.07064"},"observation_digest":"sha256:67325160a679f6d13efbd58a899f9421a3654f01a14232ec3b9b11cc8a9fbe72","observation_id":"8471acb8-d7c3-46b0-a26b-e8b30fd7cbc5","resolution":{"observed_at":"2026-05-16T07:10:43.140348Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2604.08209","last_updated":"2026-04-09T13:09:40Z","snapshot_observed_at":"2026-07-06T22:57:22.475588Z","submitted_at":"2026-04-09T13:09:40Z","title":"OmniJigsaw: Enhancing Omni-Modal Reasoning via Modality-Orchestrated Reordering","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T17:45:51.528645Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2604.08209"},"observation_digest":"sha256:5e8a2d4e973a82b208da467910fc905c17fefb931b34ea5058298191d7470716","observation_id":"5c781189-f558-491a-a1d5-8b5e42eafa9b","resolution":{"observed_at":"2026-05-11T06:11:01.948269Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2604.12527","last_updated":"2026-07-03T08:49:19Z","snapshot_observed_at":"2026-07-12T21:18:44.451500Z","submitted_at":"2026-04-14T10:00:39Z","title":"Audio-Cogito: Towards Deep Audio Reasoning in Large Audio Language Models","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T14:10:03.707886Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2604.12527"},"observation_digest":"sha256:173127d103e8cd60a47b9b21b0fd7a1ecb5efd0146a7cd35743df6ce3cb9abb6","observation_id":"78be80f0-2027-42b8-a2c5-09f9b1782fdd","resolution":{"observed_at":"2026-05-10T14:10:28.324307Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-07-12T21:18:46.566338Z","title":"Mmar: A challenging bench- mark for deep reasoning in speech, audio, music, and their mix,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.12527","last_updated":"2026-07-03T08:49:19Z","snapshot_observed_at":"2026-07-12T21:18:44.451500Z","submitted_at":"2026-04-14T10:00:39Z","title":"Audio-Cogito: Towards Deep Audio Reasoning in Large Audio Language Models","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-12T21:18:46.566338Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2604.12527"},"observation_digest":"sha256:f8a5d615d7d4f589e16cd2f5da8175a0928e2d0b23a7d60002e88a7e08e717f5","observation_id":"bd1e90bb-740a-4a33-983b-3a7427757246","resolution":{"observed_at":"2026-07-12T21:18:46.566338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2604.15804","last_updated":"2026-04-21T03:35:14Z","snapshot_observed_at":"2026-08-01T22:56:50.755050Z","submitted_at":"2026-04-17T08:05:46Z","title":"Qwen3.5-Omni Technical Report","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T08:11:22.402552Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2604.15804"},"observation_digest":"sha256:5f995db6d596c6208989e5e9201babf353b09be2addf02e49374ba11f454b13c","observation_id":"14ffd3cf-7f30-4117-b8fc-8452e52c17f8","resolution":{"observed_at":"2026-05-10T08:12:25.528544Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2604.25591","last_updated":"2026-04-28T12:56:22Z","snapshot_observed_at":"2026-08-05T12:06:36.882921Z","submitted_at":"2026-04-28T12:56:22Z","title":"Walking Through Uncertainty: An Empirical Study of Uncertainty Estimation for Audio-Aware Large Language Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-07T14:29:18.348031Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2604.25591"},"observation_digest":"sha256:551e71a9ba2810465aeb44ac0824b8b957a452c6571fb8d1e64c447a52a46458","observation_id":"d36301f1-b210-4179-b7d9-79d79e33735f","resolution":{"observed_at":"2026-05-12T00:41:26.472451Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2605.06631","last_updated":"2026-05-07T17:43:57Z","snapshot_observed_at":"2026-07-06T23:19:05.764765Z","submitted_at":"2026-05-07T17:43:57Z","title":"Task-Aware Answer Preservation under Audio Compression for Large Audio Language Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-08T03:37:54.477279Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2605.06631"},"observation_digest":"sha256:6f5f32798e954b697328ac2a7d9b802ab2b01879953410a2273b77ee585d40df","observation_id":"4a92f963-fa4a-4ffd-bede-1c716414a507","resolution":{"observed_at":"2026-05-11T22:01:11.809837Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2605.12036","last_updated":"2026-05-12T12:19:33Z","snapshot_observed_at":"2026-08-02T08:39:14.773302Z","submitted_at":"2026-05-12T12:19:33Z","title":"Towards Fine-Grained Multi-Dimensional Speech Understanding: Data Pipeline, Benchmark, and Model","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-13T04:03:27.608638Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2605.12036"},"observation_digest":"sha256:b55fdf2e284512f0d3a34166b7e0b6926591ae1af3f017371a995ec40a11cb52","observation_id":"cc39da04-14be-4fa5-b2a3-352957fc7b32","resolution":{"observed_at":"2026-05-13T04:07:13.497793Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2605.14607","last_updated":"2026-05-14T09:23:59Z","snapshot_observed_at":"2026-08-03T02:20:03.962475Z","submitted_at":"2026-05-14T09:23:59Z","title":"ViMU: Benchmarking Video Metaphorical Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-15T04:51:28.288476Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2605.14607"},"observation_digest":"sha256:08120395450611864fe2273e52b3f6953770f99428393d5b03c4cab9f7ed703e","observation_id":"91a4b310-96d0-4f1a-88dd-ba250d6a20d8","resolution":{"observed_at":"2026-05-15T04:55:04.098583Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2605.20266","last_updated":"2026-05-18T20:21:32Z","snapshot_observed_at":"2026-08-03T05:12:45.221230Z","submitted_at":"2026-05-18T20:21:32Z","title":"A Survey of Large Audio Language Models: Generalization, Trustworthiness, and Outlook","version":1},"reference_index":184,"source":"pdf_text","source_observed_at":"2026-05-21T07:38:23.099479Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2605.20266"},"observation_digest":"sha256:6fdb5e2cf138d151ccddc075e32c1c6dbb52c949e4d1aa9a14469d3fe4291744","observation_id":"dff1ec23-b6eb-4dae-99cf-683e73bdfa0e","resolution":{"observed_at":"2026-05-21T07:39:48.796691Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2605.21008","last_updated":"2026-05-20T10:44:56Z","snapshot_observed_at":"2026-07-06T23:31:32.577242Z","submitted_at":"2026-05-20T10:44:56Z","title":"A Survey of Audio Reasoning in Multimodal Foundation Models","version":1},"reference_index":121,"source":"pdf_text","source_observed_at":"2026-05-21T02:08:06.976461Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2605.21008"},"observation_digest":"sha256:7b1de03d170d9639be0390eab9286ff51934b24a68b8281b06b504e46b48f03f","observation_id":"ed348d9c-01ba-4551-b5f0-89b367c0933b","resolution":{"observed_at":"2026-05-21T02:09:24.122241Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2605.23954","last_updated":"2026-05-11T06:30:25Z","snapshot_observed_at":"2026-07-06T23:34:03.934151Z","submitted_at":"2026-05-11T06:30:25Z","title":"EchoDistill:Alignment Noisy-to-Clean Self-Distillation for Robust Audio LLMs","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-30T22:52:43.396119Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2605.23954"},"observation_digest":"sha256:b9f517c2da6f9b339ecaec3bfef998004a78eb8f59c6d39b40e42058f90bc3cf","observation_id":"3946f81a-ca8f-4bb7-9500-2be86240a790","resolution":{"observed_at":"2026-07-01T13:45:45.525920Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2605.27741","last_updated":"2026-05-26T22:34:03Z","snapshot_observed_at":"2026-08-05T02:00:17.275581Z","submitted_at":"2026-05-26T22:34:03Z","title":"Escape the Language Prior: Mitigating Late-Stage Modality Collapse in Audio Reasoning via Modality-Aware Policy Optimization","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-29T17:45:10.339950Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2605.27741"},"observation_digest":"sha256:38773b71e261e2288da8369053a8b035186a530efc66339c82a23264148fa452","observation_id":"106a13cf-e029-45a9-8dbc-0b8c15f6618b","resolution":{"observed_at":"2026-06-29T17:53:47.543232Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2605.28480","last_updated":"2026-05-27T13:39:14Z","snapshot_observed_at":"2026-08-05T07:57:19.273326Z","submitted_at":"2026-05-27T13:39:14Z","title":"Audio-Mind: An Auditable Agentic Framework for Audio Understanding","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-06-29T10:03:54.653164Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2605.28480"},"observation_digest":"sha256:0657689052153424a84040e87a8a4729980c09817fd8ac6e8c1bf16e82bf6c75","observation_id":"f915d858-d9f7-49f3-9296-81babc305094","resolution":{"observed_at":"2026-06-29T10:13:17.578490Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2605.30899","last_updated":"2026-05-29T06:33:36Z","snapshot_observed_at":"2026-08-04T13:19:05.357869Z","submitted_at":"2026-05-29T06:33:36Z","title":"A Unified and Reproducible Experimentation Framework for Speech Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-28T21:20:16.428207Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2605.30899"},"observation_digest":"sha256:1153a00c7eea42b2d66da74307770b47301feb751a90a04bbfec3518a73dcf80","observation_id":"69d7a8f7-47e5-4d33-b6c2-5d1e157bf6b8","resolution":{"observed_at":"2026-07-01T20:16:12.221545Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2606.01016","last_updated":"2026-05-31T05:13:32Z","snapshot_observed_at":"2026-07-06T23:41:39.172310Z","submitted_at":"2026-05-31T05:13:32Z","title":"PolySpeech-100: A Large-Scale Benchmark for Speech Understanding Across 100+ Languages and Dialects","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-28T17:44:07.669223Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2606.01016"},"observation_digest":"sha256:ccc84c9ca86d7bac38033800877045aea86f707f2b0353abc2cdecde7e8f5b6e","observation_id":"09b4c3d6-a41a-4c77-866a-a0fc1bb40cae","resolution":{"observed_at":"2026-06-28T17:52:27.046974Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2606.01802","last_updated":"2026-06-05T13:33:35Z","snapshot_observed_at":"2026-08-06T05:21:57.539739Z","submitted_at":"2026-06-01T07:19:22Z","title":"MOSS-Audio Technical Report","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-28T13:05:29.813707Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2606.01802"},"observation_digest":"sha256:444e0d09c7d3c845434e37cfb003c122a1ed83f1b9d6ee1c40324cefabdc59ae","observation_id":"c1785c1d-8c56-4343-a04d-f990d3fcfe49","resolution":{"observed_at":"2026-07-02T00:56:25.042011Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2606.07264","last_updated":"2026-06-05T13:39:39Z","snapshot_observed_at":"2026-07-06T23:46:55.970997Z","submitted_at":"2026-06-05T13:39:39Z","title":"VISA: A Visual Information Strengthened Audio-Reasoning System for the Interspeech 2026 ARC Agent Track","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-27T21:02:19.300441Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2606.07264"},"observation_digest":"sha256:a2c18df854dd2bbaf5e5b39b97797567b4db1e1b85a26e61ceb06aef0b28d914","observation_id":"332d5daa-7dc9-4bb5-82d2-4df03117a0d1","resolution":{"observed_at":"2026-07-02T20:07:21.343611Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2606.08425","last_updated":"2026-06-20T19:16:20Z","snapshot_observed_at":"2026-08-06T01:18:08.801931Z","submitted_at":"2026-06-07T02:50:24Z","title":"TinyGiantALM: A Compact Audio-Language Model for Intent-Aware Reasoning under Resource Constraints","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-27T18:18:41.816342Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2606.08425"},"observation_digest":"sha256:357f8e296bbc4c2a71497c6d47fef3c65f99512f52ab984ebb9301ccb4810637","observation_id":"999708c9-b3b1-45c6-9cc8-a69c06fd7859","resolution":{"observed_at":"2026-07-02T23:17:29.792398Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2606.11260","last_updated":"2026-06-09T02:38:17Z","snapshot_observed_at":"2026-08-06T14:43:05.763264Z","submitted_at":"2026-06-09T02:38:17Z","title":"RAIL: Rethinking Auditory Intelligence in Large Audio-Language Models with a CHC-Grounded Benchmark","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T12:11:56.629002Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2606.11260"},"observation_digest":"sha256:fa709672b377fa8f650ef176018565bfbc3dc17fc9c16fa08640cf218cd33514","observation_id":"ecd17084-6b4e-461d-8879-8a95dce18a54","resolution":{"observed_at":"2026-07-03T07:17:43.948271Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2606.17417","last_updated":"2026-06-16T01:57:56Z","snapshot_observed_at":"2026-07-06T23:53:02.028020Z","submitted_at":"2026-06-16T01:57:56Z","title":"A Closer Look at Failure Modes in Temporal Understanding of Large Audio-Language Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-26T23:25:46.349380Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2606.17417"},"observation_digest":"sha256:2aa3adea0940d8b074ca316c0b2cd99700994d507630ea7cba02903ec82acdda","observation_id":"d864592e-ca3e-4336-b532-a45832ade299","resolution":{"observed_at":"2026-07-03T22:39:01.636822Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2606.18273","last_updated":"2026-06-05T11:38:30Z","snapshot_observed_at":"2026-07-06T23:53:45.117607Z","submitted_at":"2026-06-05T11:38:30Z","title":"Continuous Audio Thinking for Large Audio Language Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-27T21:52:49.839901Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2606.18273"},"observation_digest":"sha256:d553dfa6c2e157f9da2b51c511c544d5361473fda464446243baab205ad6373a","observation_id":"1d713cd4-f323-4f0e-af45-1d2104b5379d","resolution":{"observed_at":"2026-07-02T17:47:17.982073Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2606.25391","last_updated":"2026-06-24T04:42:57Z","snapshot_observed_at":"2026-08-04T17:58:05.494359Z","submitted_at":"2026-06-24T04:42:57Z","title":"From Sounds to Scenes: A Benchmark for Evaluating Context-Aware Auditory Scene Understanding in Large Audio Language Models","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-06-25T20:36:24.901454Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2606.25391"},"observation_digest":"sha256:88a8cc889b1553354c079b3ac82458fd67b7e3a60e6cff26e13e1a6f1e781f02","observation_id":"5623885a-4d30-45b6-94ec-e4999e929b02","resolution":{"observed_at":"2026-07-04T20:10:07.956816Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":"2505.13032","doi":"10.48550/arxiv.2505.13032","metadata_source":"pith","pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix","venue":"cs.SD","work_id":"fe97a573-dedf-4e07-af07-b22508260a10","year":2025},"citing_paper":{"arxiv_id":"2607.05196","last_updated":"2026-07-07T15:36:48Z","snapshot_observed_at":"2026-07-11T07:46:45.098798Z","submitted_at":"2026-07-06T15:11:57Z","title":"Unified Audio Intelligence Without Regressing on Text Intelligence","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-07-07T23:59:38.702609Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2607.05196"},"observation_digest":"sha256:105f1b773e58456f327ddafebfee700f2c2eda66fb5f8894738bc02670229b3e","observation_id":"4c716412-4af6-47e6-85a1-ce211034605b","resolution":{"observed_at":"2026-07-08T00:04:22.548282Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-07-11T07:46:49.059192Z","title":"arXiv preprint arXiv:2505.13032 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.05196","last_updated":"2026-07-07T15:36:48Z","snapshot_observed_at":"2026-07-11T07:46:45.098798Z","submitted_at":"2026-07-06T15:11:57Z","title":"Unified Audio Intelligence Without Regressing on Text Intelligence","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-07-11T07:46:49.059192Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2607.05196"},"observation_digest":"sha256:697ae043d28ca0333a8a5ad652007c876a11f7d8a4e39599857f804c6fcf2e50","observation_id":"7a251b6c-e59c-4191-87b0-26578b22e099","resolution":{"observed_at":"2026-07-11T07:46:49.059192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-07-14T12:48:58.688011Z","title":"arXiv preprint arXiv:2505.13032 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.10299","last_updated":"2026-07-11T13:04:24Z","snapshot_observed_at":"2026-08-06T08:13:12.036559Z","submitted_at":"2026-07-11T13:04:24Z","title":"Empowering Long-form Omni-modal Understanding with Robust Audio Perception","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-14T12:48:58.688011Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2607.10299"},"observation_digest":"sha256:e526d478ca26d6c5442e55ce9eab36a9b86885e5aa20de10b54e2adf0dd42950","observation_id":"7186ecf6-6145-494f-9dab-65ba9f74b0a5","resolution":{"observed_at":"2026-07-14T12:48:58.688011Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-02T05:20:02.704210Z","title":"Mmar: A challenging benchmark for deep reasoning in speech, audio, music, and their mix,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13408","last_updated":"2026-07-26T03:09:03Z","snapshot_observed_at":"2026-08-05T04:45:50.000663Z","submitted_at":"2026-07-15T03:13:06Z","title":"Improving Text-to-Audio Instruction Following via Fine-Grained Feedback from Audio-Aware Large Language Models","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-02T05:20:02.704210Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2607.13408"},"observation_digest":"sha256:3025599fae6648f315cc5007aac9e99ed960b4c2c30797102e568f2f132087c8","observation_id":"845f6050-4f64-412b-ab73-8ac9afaa3b9a","resolution":{"observed_at":"2026-08-02T05:20:02.704210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-01T14:37:31.517268Z","title":"MMAR: A challenging benchmark for deep reasoning in speech, audio, music, and their mix,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.18718","last_updated":"2026-07-21T05:17:31Z","snapshot_observed_at":"2026-08-06T11:33:36.189948Z","submitted_at":"2026-07-21T05:17:31Z","title":"Summary of DCASE 2026 Task 5: Audio-Dependent Question Answering","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-01T14:37:31.517268Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2607.18718"},"observation_digest":"sha256:adf8192b0b128d4bb67e221c8218df2484ee2363760e1cf01cb5d84d87c63cbd","observation_id":"ee5013f1-6efc-402a-b10f-21f3f7de974c","resolution":{"observed_at":"2026-08-01T14:37:31.517268Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-01T07:12:17.865288Z","title":"arXiv preprint arXiv:2505.13032 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21550","last_updated":"2026-07-23T17:35:20Z","snapshot_observed_at":"2026-08-03T11:55:30.881091Z","submitted_at":"2026-07-23T17:35:20Z","title":"X$^3$-OPD: Distilling Reasoning into Large Audio-Language Models via On-Policy Alignment","version":1},"reference_index":212,"source":"arxiv_source","source_observed_at":"2026-08-01T07:12:17.865288Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2607.21550"},"observation_digest":"sha256:78f90f6b7d799b125ca8006ae0eaf8275e283ce1ad2974d1e156c839d6601ab4","observation_id":"79395fd8-cb8f-4eb4-a852-a748d76f965b","resolution":{"observed_at":"2026-08-01T07:12:17.865288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-01T00:26:31.831014Z","title":"arXiv preprint arXiv:2505.13032 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.26246","last_updated":"2026-07-28T20:31:48Z","snapshot_observed_at":"2026-08-03T10:50:56.850549Z","submitted_at":"2026-07-28T20:31:48Z","title":"Weak-to-Strong On-Policy Distillation","version":1},"reference_index":143,"source":"arxiv_source","source_observed_at":"2026-08-01T00:26:31.831014Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2607.26246"},"observation_digest":"sha256:aceb711d233f448b6673ac39fb27bc4d5a21758cedf9585040d2ab28a401dbab","observation_id":"3e2e166e-f451-4bf5-9455-58dc409ab545","resolution":{"observed_at":"2026-08-01T00:26:31.831014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13032","snapshot_observed_at":"2026-08-04T19:02:41.013510Z","title":"Maben, L","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.01881","last_updated":"2026-08-03T08:24:54Z","snapshot_observed_at":"2026-08-06T19:27:43.682503Z","submitted_at":"2026-08-03T08:24:54Z","title":"Hear, Invoke, and Understand: A Skill-Calling Multimodal Agent for Large Audio Language Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-04T19:02:41.013510Z"},"links":{"cited_paper":"/paper/2505.13032","citing_paper":"/paper/2608.01881"},"observation_digest":"sha256:0507277f60d9e4dc7066c713de0a927b27bbd77c5e483a08252f67e618f870dd","observation_id":"2e2a3dbc-9498-453c-a23b-34a3e2ccf11a","resolution":{"observed_at":"2026-08-04T19:02:41.013510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.13032/citation-record","integrity":"/paper/2505.13032/integrity","json":"/paper/2505.13032/citation-record.json","paper":"/paper/2505.13032"},"outbound":[],"paper":{"arxiv_id":"2505.13032","last_updated":"2025-05-19T12:18:42Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-04T21:45:55.994640Z","submitted_at":"2025-05-19T12:18:42Z","title":"MMAR: A Challenging Benchmark for Deep Reasoning in Speech, Audio, Music, and Their Mix"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 40 inbound Pith citation observations for arXiv:2505.13032."}