{"as_of":"2026-08-23T22:48:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7bc477fd7ec9f237a701f46af4f25550b47db5176026491188853416889ce78a","coverage":[{"denominator":54,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":54,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T20:21:32.396611Z","state":"measured"},{"denominator":61,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":61,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":7,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":7,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T20:21:32.201734Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-08T19:35:32.876839Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13237","snapshot_observed_at":"2026-08-15T20:21:32.201734Z","title":"Avg”, “Std","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.201734Z"},"links":{"cited_paper":"/paper/2505.13237","citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:1db0ba264f74c768e9e592f9d4e97487846611365248a64305295f9515376d98","observation_id":"4c13b9ac-25c0-4f0a-bd63-731527226b1c","resolution":{"observed_at":"2026-08-15T20:21:32.201734Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13237","snapshot_observed_at":"2026-08-07T05:44:06.889837Z","title":"Sakura: On the multi-hop reasoning of large audio-language models based on speech and audio information,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.07233","last_updated":"2025-09-12T18:49:28Z","snapshot_observed_at":"2026-08-18T02:56:33.217331Z","submitted_at":"2025-06-08T17:36:50Z","title":"Reducing Object Hallucination in Large Audio-Language Models via Audio-Aware Decoding","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T05:44:06.889837Z"},"links":{"cited_paper":"/paper/2505.13237","citing_paper":"/paper/2506.07233"},"observation_digest":"sha256:c1ff4851a351681a8e61acb787b0d53b7416874de90f41316de3efc97d1549a3","observation_id":"3d7972e2-000d-4c3b-9f50-2eb310168e9e","resolution":{"observed_at":"2026-08-07T05:44:06.889837Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"cited_work":{"arxiv_id":"2505.13237","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.13237","snapshot_observed_at":"2026-07-08T19:35:32.876839Z","title":"Sakura: On the multi-hop reasoning of large audio-language models based on speech and audio information","venue":"eess.AS","work_id":"548012c6-79f3-4338-9c8f-5e7df1026e7d","year":2025},"citing_paper":{"arxiv_id":"2605.20266","last_updated":"2026-05-18T20:21:32Z","snapshot_observed_at":"2026-08-16T00:00:14.809961Z","submitted_at":"2026-05-18T20:21:32Z","title":"A Survey of Large Audio Language Models: Generalization, Trustworthiness, and Outlook","version":1},"reference_index":185,"source":"pdf_text","source_observed_at":"2026-05-21T07:38:23.099479Z"},"links":{"cited_paper":"/paper/2505.13237","citing_paper":"/paper/2605.20266"},"observation_digest":"sha256:c7f463a6d40d9bb13f5f05fc594fd9b195732a0881a5b756f9af3915636865bf","observation_id":"fe0d912e-77d4-477e-a30b-7c271789c230","resolution":{"observed_at":"2026-05-21T07:39:49.074599Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"cited_work":{"arxiv_id":"2505.13237","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.13237","snapshot_observed_at":"2026-07-08T19:35:32.876839Z","title":"Sakura: On the multi-hop reasoning of large audio-language models based on speech and audio information","venue":"eess.AS","work_id":"548012c6-79f3-4338-9c8f-5e7df1026e7d","year":2025},"citing_paper":{"arxiv_id":"2605.21008","last_updated":"2026-05-20T10:44:56Z","snapshot_observed_at":"2026-08-15T09:30:05.500365Z","submitted_at":"2026-05-20T10:44:56Z","title":"A Survey of Audio Reasoning in Multimodal Foundation Models","version":1},"reference_index":131,"source":"pdf_text","source_observed_at":"2026-05-21T02:08:06.976461Z"},"links":{"cited_paper":"/paper/2505.13237","citing_paper":"/paper/2605.21008"},"observation_digest":"sha256:5d43bf64afaff4ee96ecd7e93a151598cd3962fb4aa3ebda3005ec7674db30a3","observation_id":"c8555cec-11ee-4dd0-86a2-6b22ac14f501","resolution":{"observed_at":"2026-05-21T02:09:24.095207Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"cited_work":{"arxiv_id":"2505.13237","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.13237","snapshot_observed_at":"2026-07-08T19:35:32.876839Z","title":"Sakura: On the multi-hop reasoning of large audio-language models based on speech and audio information","venue":"eess.AS","work_id":"548012c6-79f3-4338-9c8f-5e7df1026e7d","year":2025},"citing_paper":{"arxiv_id":"2606.25391","last_updated":"2026-06-24T04:42:57Z","snapshot_observed_at":"2026-08-17T07:35:57.198246Z","submitted_at":"2026-06-24T04:42:57Z","title":"From Sounds to Scenes: A Benchmark for Evaluating Context-Aware Auditory Scene Understanding in Large Audio Language Models","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-06-25T20:36:24.901454Z"},"links":{"cited_paper":"/paper/2505.13237","citing_paper":"/paper/2606.25391"},"observation_digest":"sha256:52a06865fc9d1b62c173fc1a1dbdc4d8b06545ab379be67c0d309e5e33d2dedf","observation_id":"bd865246-2e24-4a45-a29e-31ce1e8a1e8f","resolution":{"observed_at":"2026-07-04T20:10:07.947534Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"cited_work":{"arxiv_id":"2505.13237","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.13237","snapshot_observed_at":"2026-07-08T19:35:32.876839Z","title":"Sakura: On the multi-hop reasoning of large audio-language models based on speech and audio information","venue":"eess.AS","work_id":"548012c6-79f3-4338-9c8f-5e7df1026e7d","year":2025},"citing_paper":{"arxiv_id":"2607.06014","last_updated":"2026-07-07T08:57:00Z","snapshot_observed_at":"2026-08-13T18:43:43.361960Z","submitted_at":"2026-07-07T08:57:00Z","title":"Escaping the Procrustean Bed: Groupwise Orthogonal Connectors for Audio-Language Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-08T19:31:48.172439Z"},"links":{"cited_paper":"/paper/2505.13237","citing_paper":"/paper/2607.06014"},"observation_digest":"sha256:bf3ac22179ad8c5197ed04a819de47d42ad1551d5e5a7400c46a34ef95711b17","observation_id":"865f8dc1-fbc9-4c8b-8930-afa1b769520c","resolution":{"observed_at":"2026-07-08T19:35:32.879177Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13237","snapshot_observed_at":"2026-08-02T01:12:30.629444Z","title":"Sakura: On the multi-hop reasoning of large audio-language models based on speech and audio information,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14753","last_updated":"2026-07-16T09:38:45Z","snapshot_observed_at":"2026-08-14T02:41:14.466076Z","submitted_at":"2026-07-16T09:38:45Z","title":"Large Audio Language Models for Spoofing-Aware Speaker Verification","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-02T01:12:30.629444Z"},"links":{"cited_paper":"/paper/2505.13237","citing_paper":"/paper/2607.14753"},"observation_digest":"sha256:49ed0d4b8fa1f033753552b2b7a51ba6493488dffc240068f12b859d1f75a71c","observation_id":"3439d7d3-c933-4473-a804-f21bf2009162","resolution":{"observed_at":"2026-08-02T01:12:30.629444Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.13237/citation-record","integrity":"/paper/2505.13237/integrity","json":"/paper/2505.13237/citation-record.json","paper":"/paper/2505.13237"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:33.018426Z","title":"What is the animal in the sound ?","venue":null,"work_id":"604acad7-b74c-4e51-8ca7-9bd14546bfcf","year":null},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.197075Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:faaf2665d3d20f60aaad50a41c5e781c876dc6fa3703ce41691b3a25229d4fcb","observation_id":"220f6e95-2554-4342-a9f0-7f49053abf4f","resolution":{"observed_at":"2026-08-15T20:21:33.022387Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13237","snapshot_observed_at":"2026-08-15T20:21:32.201734Z","title":"Avg”, “Std","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.201734Z"},"links":{"cited_paper":"/paper/2505.13237","citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:1db0ba264f74c768e9e592f9d4e97487846611365248a64305295f9515376d98","observation_id":"4c13b9ac-25c0-4f0a-bd63-731527226b1c","resolution":{"observed_at":"2026-08-15T20:21:32.201734Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:33.007504Z","title":"determining the age of the speaker","venue":null,"work_id":"ec9e03cf-d405-4c32-aac2-c7cdd5559734","year":null},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.206065Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:be1a21a5eb389a0f7a984d93b6ea24f36bd931af582583ffc2c33ffb4058d551","observation_id":"6eccb163-3f2e-479e-9430-beed7a69738a","resolution":{"observed_at":"2026-08-15T20:21:33.011252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.996270Z","title":"the association of animals and human personality","venue":null,"work_id":"51db3e45-32f4-427b-b83d-626344f30e8d","year":null},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.210053Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:0574d8ad23c7b7ff8c9b4e603dbbd19665cadc006f65b18a96d6f2cefe215a40","observation_id":"1596be03-4632-424f-b663-cb13a911e7bf","resolution":{"observed_at":"2026-08-15T20:21:33.000596Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.984298Z","title":"feeding habits","venue":null,"work_id":"1fb9d8d6-f5ef-44de-a876-0d0e82f88682","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.214388Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:e76c49b89490173fb05c201cd6cca944449895caee5664ba7b38af2f1110a5c0","observation_id":"82f54d82-16b2-4222-89d6-f98e324e0b4f","resolution":{"observed_at":"2026-08-15T20:21:32.988792Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.972553Z","title":"Evaluation metrics Since SAKURA comprises multiple-choice questions, accu- racy is a natural metric","venue":null,"work_id":"3cf68cc1-decf-4c88-8ef8-4fb24c55542f","year":null},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.218515Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:b087c43a80aa34b5e967903fa55820141f3258820874d2524d22fdaab9e3986d","observation_id":"f2cfe7c5-c50a-40c3-b6b2-1d20fc7625a1","resolution":{"observed_at":"2026-08-15T20:21:32.976646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.961847Z","title":null,"venue":null,"work_id":"45c88dc9-d30e-4d2a-8809-eb32844887f4","year":null},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.222331Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:3f1b62fe9559561456f8c085286d296eb4f9cbe2284c576326217eccfe399dea","observation_id":"5ce3c14d-5bfe-47bf-b6a8-5a63ee1d856e","resolution":{"observed_at":"2026-08-15T20:21:32.965380Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.951373Z","title":null,"venue":null,"work_id":"2ec1ce52-f62a-4cc5-9a0f-28c551f3044a","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.226358Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:33433af2cf11d6740f4cb198ba3e53eee3442417b6f12863714a4b99b441e6b6","observation_id":"6dab4d2d-dd47-4a2f-be37-b6af1cd00e5d","resolution":{"observed_at":"2026-08-15T20:21:32.954893Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.940822Z","title":"cor- rect/incorrect","venue":null,"work_id":"087f02c6-65c2-4fef-aa43-fb0f984eddc7","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.230095Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:79ad3b4176851fc58fa8c9419edd6ebaa96e6328d3f66c8b9ca9af17c00ba7ec","observation_id":"1a8ee3e0-e0f2-4b28-aa05-d75b5b37716c","resolution":{"observed_at":"2026-08-15T20:21:32.944431Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.929201Z","title":"The animal making the sound is cat","venue":null,"work_id":"dee90c9c-6a91-493f-b41b-9eddba712c34","year":null},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.233913Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:3a6dd64fd2ea516278eb24d5fe6b398e45d0bc105cffe3e337dce7794b15d250","observation_id":"9304f2e5-aa61-4d2d-96c4-fe957780d27c","resolution":{"observed_at":"2026-08-15T20:21:32.933058Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.917412Z","title":"Our findings show that LALMs struggle to recognize certain speech and audio attributes, exhibiting perception blind spots","venue":null,"work_id":"bd6c747f-e074-401f-8bac-9407c69ddea7","year":null},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.238091Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:87c8a32127a473f66a772cafd720647cae99e055dd8b7dad359908811acd0344","observation_id":"08e88832-d58e-4cf7-984d-c1aec4d716a8","resolution":{"observed_at":"2026-08-15T20:21:32.922456Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.906274Z","title":null,"venue":null,"work_id":"855154ee-7642-47d2-8c98-4dab979453cb","year":null},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.241681Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:9aab4f60e692841b24e257c21cd52cdcece074ecf7748348fe77b109f2a095f2","observation_id":"69d419af-1c17-42e5-af17-106368558be4","resolution":{"observed_at":"2026-08-15T20:21:32.910134Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-15T20:21:32.245573Z","title":"The llama 3 herd of models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.245573Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:7416c7755e3478a1d7f6e86cc12c4addcad828ab11cd866bea8e10b00f712fca","observation_id":"df627084-5ec7-434e-a390-38670676510a","resolution":{"observed_at":"2026-08-15T20:21:32.245573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-08-15T14:02:47.366139Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-15T20:21:32.249490Z","title":"Gpt-4o system card,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.249490Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:d62799f055b30eb42878472ff285d1166843b574b17a41a39b3935c997d8150e","observation_id":"a6208b81-601c-478c-b1d4-d26fe7ea1830","resolution":{"observed_at":"2026-08-15T20:21:32.249490Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.896201Z","title":"Vipergpt: Visual inference via python execution for reasoning,","venue":null,"work_id":"1e5b741f-16b0-4799-8254-a0cb362c23a6","year":2023},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.253711Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:477ff3dfac291266808bbc593b3bcfea22f791cfc0f6f61094f409e1b3f7b225","observation_id":"76c6f9ea-5555-49d5-ace0-e900c99a6281","resolution":{"observed_at":"2026-08-15T20:21:32.899629Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.886107Z","title":"Audiogpt: Understanding and generating speech, music, sound, and talking head,","venue":null,"work_id":"f520b75f-805b-491b-b60e-18c9fe0494e8","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.257761Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:73ff93a57832edeea8397ff491252ecb6e5ec38eb3b7771adf7385a84e607add","observation_id":"584c2320-3e69-40db-864e-55828a41fcb6","resolution":{"observed_at":"2026-08-15T20:21:32.889553Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.875462Z","title":"Speech-copilot: Leveraging large language models for speech processing via task decomposition, modular- ization, and program generation,","venue":null,"work_id":"b641524d-4964-44b0-8e08-d1bb21b12088","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.261763Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:1e69d30a4f9a35c8677c7b558ee462e677e7d6af8bb8434e8598c359c82bbc9d","observation_id":"ffa26111-18b8-4810-ade7-d37de708f3e3","resolution":{"observed_at":"2026-08-15T20:21:32.879308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.864351Z","title":"Visual instruction tuning,","venue":null,"work_id":"0a202553-f843-41a9-b465-caeaa745351a","year":2023},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.265698Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:6979ce8cfd603c1e4190ec74846bdd7e2b308c70d1fd0e4e676e70576f86be43","observation_id":"18f55ae7-869c-41bd-987f-b7223f036914","resolution":{"observed_at":"2026-08-15T20:21:32.867915Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-08-14T18:15:53.516440Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-15T20:21:32.269489Z","title":"Gemini 1.5: Unlocking multimodal under- standing across millions of tokens of context,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.269489Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:4ad0c1b51981f01a749ca121e488b6d9e2b9c09bbce80c283101bcffbecb0452","observation_id":"1ea6430a-1dc2-4c33-a0ae-dfae28f764c7","resolution":{"observed_at":"2026-08-15T20:21:32.269489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.853905Z","title":"Joint audio and speech understanding,","venue":null,"work_id":"a13e18f9-63f1-47d6-b3f1-1fa8c0fde3c7","year":2023},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.273650Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:7b60dd9e8d5fa9b4a5b1ea62d24a6a3e1507b955dcf5e70f194b31541b7c1d0d","observation_id":"0ead504f-3ed8-40ec-b36b-c322eda65934","resolution":{"observed_at":"2026-08-15T20:21:32.857566Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-21T16:14:12.045334Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-15T20:21:32.277711Z","title":"Gama: A large audio-language model with advanced audio understanding and complex reasoning abilities,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.277711Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:bea6b9663cb7755601b23cdab90875aec81dbe6459365dae4ac6c62a964ca1aa","observation_id":"30b44235-586e-4815-9852-dda99f3f0037","resolution":{"observed_at":"2026-08-15T20:21:32.277711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.843339Z","title":"SALMONN: Towards generic hearing abilities for large language models,","venue":null,"work_id":"a09e5571-0f51-47d0-8094-6579dcccd040","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.282098Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:294f28df8dc6320b296df67b8715cde079fbe2aae96ce18614fbd1961f1f8d2f","observation_id":"3c815fee-4775-487b-aa1b-e3a98d4600b0","resolution":{"observed_at":"2026-08-15T20:21:32.847207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.20007","last_updated":"2025-01-27T08:46:09Z","snapshot_observed_at":"2026-08-22T06:27:55.863481Z","submitted_at":"2024-09-30T07:01:21Z","title":"DeSTA2: Developing Instruction-Following Speech Language Model Without Speech Instruction-Tuning Data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.20007","snapshot_observed_at":"2026-08-15T20:21:32.285531Z","title":"Developing instruction-following speech lan- guage model without speech instruction-tuning data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.285531Z"},"links":{"cited_paper":"/paper/2409.20007","citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:0eb03506c7415979043c9742dd30d11c840f173f552cfe3c86344dbd85531804","observation_id":"37ec7eed-b848-4949-b7bc-98b3b8f1c52a","resolution":{"observed_at":"2026-08-15T20:21:32.285531Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07919","last_updated":"2023-12-21T10:20:42Z","snapshot_observed_at":"2026-08-07T10:17:55.688598Z","submitted_at":"2023-11-14T05:34:50Z","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07919","snapshot_observed_at":"2026-08-15T20:21:32.289365Z","title":"Qwen-audio: Advancing universal audio under- standing via unified large-scale audio-language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.289365Z"},"links":{"cited_paper":"/paper/2311.07919","citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:74e72e2590b02b1d3a852f723037be0e081c2843f1c7b9295ee27153db81d968","observation_id":"bec010cc-24e9-4668-a73d-521798b53e67","resolution":{"observed_at":"2026-08-15T20:21:32.289365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10759","last_updated":"2024-07-15T14:38:09Z","snapshot_observed_at":"2026-08-14T01:27:16.843576Z","submitted_at":"2024-07-15T14:38:09Z","title":"Qwen2-Audio Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10759","snapshot_observed_at":"2026-08-15T20:21:32.293038Z","title":"Qwen2-audio technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.293038Z"},"links":{"cited_paper":"/paper/2407.10759","citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:698c95e7aa4fb0d216969992c60205ff6078dc90c8a969b623e28379c5f5a137","observation_id":"ed537ded-5d6c-4063-8574-f63aa56d655c","resolution":{"observed_at":"2026-08-15T20:21:32.293038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.831972Z","title":"A peek into token bias: Large language models are not yet genuine reasoners,","venue":null,"work_id":"32743c20-9153-4e28-9dc3-9320e7690a29","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.296768Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:36c460b6e3f4748efbc09f4190ab4106b35024a896e3da3a8f20a27efea838b3","observation_id":"acea8f48-0190-4efc-a583-0b575169e643","resolution":{"observed_at":"2026-08-15T20:21:32.835964Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.820382Z","title":"Large language models cannot self-correct rea- soning yet,","venue":null,"work_id":"5d770bd8-a3b6-4948-be64-6a632b51805e","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.300368Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:46b4a586627fb3044f244a93424a0cb4667dfb95de9ac1099c2c9b36e9492614","observation_id":"1ca49faf-2d01-4ddc-bd12-72af4a3160fa","resolution":{"observed_at":"2026-08-15T20:21:32.824357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.810081Z","title":"Premise order matters in reasoning with large language models,","venue":null,"work_id":"cd1e9904-432d-441a-9f3c-fe78d91cd4ec","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.303863Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:8a107fead01bdc041b95e89570143a12f4602952f3a2d13ba89a6377368a9e49","observation_id":"4ce53e49-17c5-42aa-baec-6e1d331cd2b6","resolution":{"observed_at":"2026-08-15T20:21:32.813640Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.799370Z","title":"Measuring and narrowing the compositionality gap in language models,","venue":null,"work_id":"6e81cb7d-382f-4d5a-9a2e-16034582969a","year":2023},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.307710Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:9f2d8c7b5791a5b400a8992f6a6f458af1227120b7cbec28435e2089d10dbcaf","observation_id":"fd902063-b37b-4816-8200-0e012c9486ff","resolution":{"observed_at":"2026-08-15T20:21:32.802918Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13858","last_updated":"2024-06-19T21:36:40Z","snapshot_observed_at":"2026-08-22T13:20:38.398658Z","submitted_at":"2024-06-19T21:36:40Z","title":"Distributional reasoning in LLMs: Parallel reasoning processes in multi-hop reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13858","snapshot_observed_at":"2026-08-15T20:21:32.311200Z","title":"Distributional reasoning in llms: Parallel reasoning processes in multi-hop reasoning,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.311200Z"},"links":{"cited_paper":"/paper/2406.13858","citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:dc76f4c30c5d17ea7c362c1766f65c6b49348525bc63b28a4d508dccaaaf7a94","observation_id":"f55d04c0-601c-471e-8634-704944054c43","resolution":{"observed_at":"2026-08-15T20:21:32.311200Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.788886Z","title":"Do large language models latently perform multi- hop reasoning?","venue":null,"work_id":"c9b6cdc4-ca95-4a1e-9c92-27a55c2e0b0f","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.315001Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:b78875f3ca46f6311c1c6fef20078f10986cd8c8a531cb3292c37365630bb54c","observation_id":"14ebca65-3274-46b8-8230-00a8ca8696a1","resolution":{"observed_at":"2026-08-15T20:21:32.792689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.778639Z","title":"Hopping too late: Exploring the limitations of large language models on multi-hop queries,","venue":null,"work_id":"daf1f601-15f7-403e-91b3-21aeb709d30b","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.318751Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:97e0b3d9ce6a1737054f2e30f31432e3837b0576c65bc895115a2431904f5029","observation_id":"dc7deb39-27c9-4983-a7a3-28c91a14cc9f","resolution":{"observed_at":"2026-08-15T20:21:32.782218Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.767843Z","title":"Investigating multi-hop factual shortcuts in knowl- edge editing of large language models,","venue":null,"work_id":"fd2b1b2a-2f2a-4479-8e20-5416fe5c9a1e","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.322290Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:af8af7eecc3ab95167bfdd7047c92a3fc357bc20442fa24df2f53a6333d0d7fe","observation_id":"690ccb98-c203-4867-92e7-db237b0abe5d","resolution":{"observed_at":"2026-08-15T20:21:32.771545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.756645Z","title":"Dynamic-SUPERB phase-2: A collabora- tively expanding benchmark for measuring the capabilities of spo- ken language models with 180 tasks,","venue":null,"work_id":"be625c0d-ccd9-4b8e-be4b-fbeb8896e0b1","year":2025},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.325857Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:b7eac9cc72e8747dc15dbcd9733863c21977336b5ea56a533ec90baf937d52da","observation_id":"4f74e161-5e76-42ca-8120-bf4935db7ab3","resolution":{"observed_at":"2026-08-15T20:21:32.760585Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.744913Z","title":"AIR-bench: Benchmarking large audio-language models via generative comprehension,","venue":null,"work_id":"c64f2884-b30c-4418-bd16-a32530a15b5b","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.329530Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:54e653317e7859e153d315a18ba2282d91dd66d3a2e386ef7545ca34162288d6","observation_id":"ff81b993-a8ad-474c-a1c7-2d8a2c985df9","resolution":{"observed_at":"2026-08-15T20:21:32.749575Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.734046Z","title":"Advancing large lan- guage models to capture varied speaking styles and respond prop- erly in spoken conversations,","venue":null,"work_id":"b46340f2-d8a5-4374-8dc8-f70ef5dc29ba","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.333173Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:ff256f0e7500a18223c68a052fd608950a289b998e89bce6e3aa1b3482adab96","observation_id":"9152f3d8-54dd-47f5-a434-86ff5ecfad57","resolution":{"observed_at":"2026-08-15T20:21:32.738002Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.722196Z","title":"Sd-eval: A benchmark dataset for spoken dialogue understanding beyond words,","venue":null,"work_id":"37ef0cf6-0b30-4fce-a1cc-7d380883cd83","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.337049Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:130ed89a18605c8ffef38226a08586e0716639da97be462d684925882f236b2a","observation_id":"e669ad24-3ff2-44b0-81f0-4f90e32575e1","resolution":{"observed_at":"2026-08-15T20:21:32.726406Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.709104Z","title":"Listen and speak fairly: a study on semantic gender bias in speech integrated large language models,","venue":null,"work_id":"818b5080-0f2b-4a4c-b674-6b4c791d98ff","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.340670Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:20f3ad5c09b23190bff9a67b0cedae29d25fdeed95b90c7ae7fd33f9730b3fd4","observation_id":"5ef16883-0dfa-490d-9617-d2c5201cb568","resolution":{"observed_at":"2026-08-15T20:21:32.713188Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.696696Z","title":"Spoken stereoset: on eval- uating social bias toward speaker in speech large language mod- els,","venue":null,"work_id":"9d07c8ce-77f1-4fbc-8244-e2122750bcf9","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.344133Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:7e9195c2a9e8cf27f399f015c46b31d442ae42f9cb5f4b67384e3363b6d4a3c7","observation_id":"3edb1952-9549-45e0-856e-5ed3977b9377","resolution":{"observed_at":"2026-08-15T20:21:32.700574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.685331Z","title":"Compa: Addressing the gap in compositional reasoning in audio-language models,","venue":null,"work_id":"fa0abb04-2364-4793-ac31-58022b191653","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.347722Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:9652bbb9aa2d61fb35fd89f9036ca7c388268ca3d9d4c6a8af7390a2c5e8af0b","observation_id":"08bf72d3-b129-4616-9cc0-4656b2e22f82","resolution":{"observed_at":"2026-08-15T20:21:32.689090Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.673584Z","title":"MMAU: A massive multi-task audio understand- ing and reasoning benchmark,","venue":null,"work_id":"de3ac3cf-f0ee-4cbb-bf4e-07da8a9db427","year":2025},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.351070Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:8b7ec931409ada991a04571fffd2416167d8e7017ce6c999a9a42badaf0a5fc9","observation_id":"b35f9a55-78e8-48fb-bec1-304d13c7d394","resolution":{"observed_at":"2026-08-15T20:21:32.677756Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.661087Z","title":"Understanding sounds, missing the questions: The challenge of object hallucination in large audio-language models,","venue":null,"work_id":"fd76a8e5-4b9a-4381-b8df-211595e7fa18","year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.354348Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:cb043dde58f4fffa363d4a42688220b1e1b25aac26807eb066579512136c39f6","observation_id":"a33989dd-b7a6-4dd9-932a-7e3f33601be1","resolution":{"observed_at":"2026-08-15T20:21:32.664808Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.649481Z","title":"Common voice: A massively-multilingual speech corpus,","venue":null,"work_id":"21aa36e6-7a24-40f4-b324-b071bdc35cf6","year":2020},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.357698Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:ef2e39da7c9bf5bda312dc48ae8157e7d43a86e9ba7e84b88f3bfffb17f5e3c2","observation_id":"f02676a3-5c95-494c-aa50-a5f3ec4c02b0","resolution":{"observed_at":"2026-08-15T20:21:32.653515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.637736Z","title":"Crema-d: Crowd-sourced emotional multimodal actors dataset,","venue":null,"work_id":"9ed47469-d8f9-4894-8944-dbac8c71397e","year":2014},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.361081Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:38a6afe7af3e7a66783339ff40e005be9b634b8763e4fff1309269cd773360ce","observation_id":"d3c3a2db-b78c-4f8a-852e-ccb507698131","resolution":{"observed_at":"2026-08-15T20:21:32.641895Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.624563Z","title":"MELD: A multimodal multi-party dataset for emotion recognition in conversations,","venue":null,"work_id":"a27ae01e-cd07-4da0-89c5-a99994f568e2","year":2019},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.364575Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:4c1ed3fac1c3fd634f72ed92e42418e22627bc4faf260645e8fbdaa79e3bea78","observation_id":"cd4425f2-73e1-4248-b773-d9c702ec0d21","resolution":{"observed_at":"2026-08-15T20:21:32.628960Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.610730Z","title":"Emotion detection on tv show transcripts with sequence-based convolutional neural networks,","venue":null,"work_id":"7c8541df-7839-4b5f-85c8-ab6cdc375f47","year":2018},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.367947Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:23d69dcc95d764da5b3e7349bc15141160c0b5c901bec4c29525d921db4d35cb","observation_id":"15e44130-5b1e-46d9-8302-bf8b0ba04df6","resolution":{"observed_at":"2026-08-15T20:21:32.615043Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.371393Z","title":"Esc: Dataset for environmental sound classifica- tion,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.371393Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:82cf926f3a86f68ae3e895bcb50c5c13138a0148de82415930a819beca4143cb","observation_id":"e9d44583-5f75-4195-b726-08e7658b512b","resolution":{"observed_at":"2026-08-15T20:21:32.371393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.591421Z","title":"Animal sound classification using a convolutional neural network,","venue":null,"work_id":"6120cf43-a532-4b8b-aeae-5749969607d1","year":2018},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.374940Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:a9f87f2a48d290da0b2bfc35b1019a3f4d6ee3484267adc9a641fac99dda281b","observation_id":"1afc1e41-2823-425b-9693-79406bc848b9","resolution":{"observed_at":"2026-08-15T20:21:32.595626Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15594","last_updated":"2025-10-19T10:32:43Z","snapshot_observed_at":"2026-08-20T12:52:29.141935Z","submitted_at":"2024-11-23T16:03:35Z","title":"A Survey on LLM-as-a-Judge","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15594","snapshot_observed_at":"2026-08-15T20:21:32.378192Z","title":"A survey on llm-as-a-judge,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.378192Z"},"links":{"cited_paper":"/paper/2411.15594","citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:38d6989c2536d62789046fe15277df23e0b699685af77d514cb8b95a629380f1","observation_id":"32077ba6-b845-44b7-95f0-8bb3e4eb089d","resolution":{"observed_at":"2026-08-15T20:21:32.378192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:32.576298Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":"40c3e589-75e3-4a07-9ca6-22c014ae38ab","year":2023},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.381882Z"},"links":{"citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:b89ba4500f382269211b274618013bcda4d8f218319e11ce0d0a7f086b229e22","observation_id":"7e60d6e0-1752-4fc7-8427-acc0c7eef80a","resolution":{"observed_at":"2026-08-15T20:21:32.582671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.09940","last_updated":"2025-02-14T06:34:08Z","snapshot_observed_at":"2026-08-21T04:48:15.247350Z","submitted_at":"2025-02-14T06:34:08Z","title":"A Preliminary Exploration with GPT-4o Voice Mode","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.09940","snapshot_observed_at":"2026-08-15T20:21:32.385350Z","title":"A preliminary exploration with gpt-4o voice mode,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.385350Z"},"links":{"cited_paper":"/paper/2502.09940","citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:3596402fc634bddac7436f289e93395f98cb69dfa1a2a4a6466f4d8cdeacbdb0","observation_id":"3388e39a-3943-45a1-bfc0-4c41da4c7261","resolution":{"observed_at":"2026-08-15T20:21:32.385350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.06666","last_updated":"2025-03-01T12:59:49Z","snapshot_observed_at":"2026-08-16T13:19:40.907971Z","submitted_at":"2024-09-10T17:34:34Z","title":"LLaMA-Omni: Seamless Speech Interaction with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.06666","snapshot_observed_at":"2026-08-15T20:21:32.388918Z","title":"Llama-omni: Seamless speech interaction with large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.388918Z"},"links":{"cited_paper":"/paper/2409.06666","citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:cec7418f1fa78375dae4d17ae952c399f0dade9fae0536ae71e74cee8030c6e4","observation_id":"2be34161-d75b-4350-bbdb-f542090ae388","resolution":{"observed_at":"2026-08-15T20:21:32.388918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00037","last_updated":"2024-10-02T09:11:45Z","snapshot_observed_at":"2026-07-30T10:21:14.474746Z","submitted_at":"2024-09-17T17:55:39Z","title":"Moshi: a speech-text foundation model for real-time dialogue","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00037","snapshot_observed_at":"2026-08-15T20:21:32.393008Z","title":"Moshi: a speech-text foundation model for real-time dialogue,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.393008Z"},"links":{"cited_paper":"/paper/2410.00037","citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:09e0f6d535f5c4be9eec6f11822e4018902251a2e19fc7cb7cd00354b5d7dde0","observation_id":"44c08ff7-7763-4a7c-be55-7f4cba09cb1c","resolution":{"observed_at":"2026-08-15T20:21:32.393008Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.07111","last_updated":"2024-12-27T07:29:19Z","snapshot_observed_at":"2026-08-16T13:01:06.676582Z","submitted_at":"2024-11-11T16:37:40Z","title":"Building a Taiwanese Mandarin Spoken Language Model: A First Attempt","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.07111","snapshot_observed_at":"2026-08-15T20:21:32.396611Z","title":"Building a taiwanese mandarin spoken lan- guage model: A first attempt,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-15T20:21:32.396611Z"},"links":{"cited_paper":"/paper/2411.07111","citing_paper":"/paper/2505.13237"},"observation_digest":"sha256:4a9e8c030f7db82b864f494757ca39e4c88bb0d535fe3636d3095d5695de8f4b","observation_id":"f223fe9a-a743-4c20-8ed4-c5df12d9007d","resolution":{"observed_at":"2026-08-15T20:21:32.396611Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.13237","last_updated":"2025-08-24T16:09:17Z","latest_version":3,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-21T23:18:20.561146Z","submitted_at":"2025-05-19T15:20:32Z","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information"},"reference_resolution":{"displayed":54,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":18,"verified_exact":0,"verified_fuzzy":36},"total_outbound_references":54},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 54 of 54 outbound references and 7 inbound Pith citation observations for arXiv:2505.13237."}