{"as_of":"2026-08-07T08:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ff8cc9b88b1d231d00a6b1d9e1d7702e4c664e435fd3a05b158f5bb7317740ad","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":21,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":21,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":21,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":21,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:54:13.519625Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":"2406.11768","doi":"10.48550/arxiv.2406.11768","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GAMA: A large audio- language model with advanced audio understanding and complex rea- soning abilities","venue":"arXiv (Cornell University)","work_id":"f407a4af-a90d-4687-b15e-70573cc84a5d","year":2024},"citing_paper":{"arxiv_id":"2504.18425","last_updated":"2025-04-25T15:31:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-25T15:31:46Z","title":"Kimi-Audio Technical Report","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-11T19:21:26.933349Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2504.18425"},"observation_digest":"sha256:1ec05bbd68b18828659e0878559425f99b21ca23ff3b0c31a6a64e4d3a3a2870","observation_id":"6988f0d0-69cc-4727-896c-f2bdbc257da9","resolution":{"observed_at":"2026-05-11T19:21:27.293952Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":"2406.11768","doi":"10.48550/arxiv.2406.11768","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GAMA: A large audio- language model with advanced audio understanding and complex rea- soning abilities","venue":"arXiv (Cornell University)","work_id":"f407a4af-a90d-4687-b15e-70573cc84a5d","year":2024},"citing_paper":{"arxiv_id":"2505.16933","last_updated":"2025-06-04T05:52:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-22T17:23:26Z","title":"LLaDA-V: Large Language Diffusion Models with Visual Instruction Tuning","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-17T03:46:06.074416Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2505.16933"},"observation_digest":"sha256:605b235fb41ed72a2364441577df66e572613a0db8aa5d80c89a2e8b95932a44","observation_id":"06d1d233-677c-4da6-a2ad-61236a8de789","resolution":{"observed_at":"2026-05-17T03:46:06.395741Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-07T04:54:13.519625Z","title":"Gama: A large audio-language model with advanced audio under- standing and complex reasoning abilities,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09375","last_updated":"2025-08-23T19:55:31Z","snapshot_observed_at":"2026-08-07T04:47:32.667457Z","submitted_at":"2025-06-11T03:50:16Z","title":"CoLMbo: Speaker Language Model for Descriptive Profiling","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T04:54:13.519625Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2506.09375"},"observation_digest":"sha256:3c3630bd2b5f93a1da9089baa13506928e3e7317bdcc0c7cef4be4d44706fe82","observation_id":"9c967c50-3543-44f4-b64c-13e5197827fb","resolution":{"observed_at":"2026-08-07T04:54:13.519625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-06T19:26:53.686731Z","title":"Gama: A large audio-language model with advanced audio un- derstanding and complex reasoning abilities,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05609","last_updated":"2025-07-08T02:37:20Z","snapshot_observed_at":"2026-08-06T19:19:37.382327Z","submitted_at":"2025-07-08T02:37:20Z","title":"MMW: Side Talk Rejection Multi-Microphone Whisper on Smart Glasses","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T19:26:53.686731Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2507.05609"},"observation_digest":"sha256:b42a782daef8ac258dc65613110e1a0280c5669380db0a35f3bca4705d543535","observation_id":"d7d42a5d-675f-4ef8-b663-e85edb03d41f","resolution":{"observed_at":"2026-08-06T19:26:53.686731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-06T19:46:11.380266Z","title":"Ghosh, S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06256","last_updated":"2025-07-07T07:29:52Z","snapshot_observed_at":"2026-08-07T01:19:10.776990Z","submitted_at":"2025-07-07T07:29:52Z","title":"Attacker's Noise Can Manipulate Your Audio-based LLM in the Real World","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T19:46:11.380266Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2507.06256"},"observation_digest":"sha256:e14888111f8711d8963f4af2ceac8a74813810dc024bcda6e1e54ba2a647ea9b","observation_id":"2e732449-e696-4d5f-ae28-3f39fa1338d7","resolution":{"observed_at":"2026-08-06T19:46:11.380266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-06T17:45:45.070152Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10016","last_updated":"2025-08-20T07:04:41Z","snapshot_observed_at":"2026-08-06T17:39:34.024953Z","submitted_at":"2025-07-14T07:51:56Z","title":"The Man Behind the Sound: Demystifying Audio Private Attribute Profiling via Multimodal Large Language Model Agents","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T17:45:45.070152Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2507.10016"},"observation_digest":"sha256:3970ae576e6645e362d1334942f84daeb873c2bc030681fd46b393250bfe1f79","observation_id":"a4ccc679-865f-47a5-9b35-54313167a882","resolution":{"observed_at":"2026-08-06T17:45:45.070152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-06T14:50:04.254194Z","title":"ArXiv abs/2406.11768 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17563","last_updated":"2025-07-23T14:53:50Z","snapshot_observed_at":"2026-08-07T03:22:50.485332Z","submitted_at":"2025-07-23T14:53:50Z","title":"BoSS: Beyond-Semantic Speech","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T14:50:04.254194Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2507.17563"},"observation_digest":"sha256:3d44e7b8f4598fb3bdc769256a23a07069e2565f7ca247ae1d6a14a936c8d8e8","observation_id":"97846861-2b18-4eba-a1bd-0d568262121a","resolution":{"observed_at":"2026-08-06T14:50:04.254194Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-05T18:23:28.522442Z","title":"Sakshi, Oriol Ni- eto, Ramani Duraiswami, and Dinesh Manocha","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.00029","last_updated":"2025-08-20T13:54:53Z","snapshot_observed_at":"2026-08-05T18:23:26.064307Z","submitted_at":"2025-08-20T13:54:53Z","title":"From Sound to Sight: Towards AI-authored Music Videos","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T18:23:28.522442Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2509.00029"},"observation_digest":"sha256:d8d290e8424cf9b5a7229a039c57f818363b940944c788eccd1d9c882683c4d6","observation_id":"b03fa92e-d669-4b4c-b3cc-618d47076bdd","resolution":{"observed_at":"2026-08-05T18:23:28.522442Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-04T17:54:57.124210Z","title":"Gama: A large audio-language model with ad- vanced audio understanding and complex reasoning abilities,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.10391","last_updated":"2025-09-12T16:31:20Z","snapshot_observed_at":"2026-08-05T08:01:36.767909Z","submitted_at":"2025-09-12T16:31:20Z","title":"Improving Audio Event Recognition with Consistency Regularization","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T17:54:57.124210Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2509.10391"},"observation_digest":"sha256:3b1ecf889141c2dc7ebbbc221355a22cf54dca25854fff0d1a32378bc438c143","observation_id":"aaef49ed-b065-46dc-bb5a-1772b68ba670","resolution":{"observed_at":"2026-08-04T17:54:57.124210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":"2406.11768","doi":"10.48550/arxiv.2406.11768","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GAMA: A large audio- language model with advanced audio understanding and complex rea- soning abilities","venue":"arXiv (Cornell University)","work_id":"f407a4af-a90d-4687-b15e-70573cc84a5d","year":2024},"citing_paper":{"arxiv_id":"2512.02231","last_updated":"2026-04-10T02:23:14Z","snapshot_observed_at":"2026-07-06T22:37:31.360740Z","submitted_at":"2025-12-01T21:57:26Z","title":"See, Hear, and Understand: Benchmarking Audiovisual Human Speech Understanding in Multimodal Large Language Models","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-17T02:12:55.170296Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2512.02231"},"observation_digest":"sha256:71b8da3b1b30c04184509dfd3411bb52af607a3b7da039a8b5878a9940e651e8","observation_id":"9ac6ba19-b231-44c4-a136-9a515157f6f3","resolution":{"observed_at":"2026-05-17T02:13:52.187373Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-03T12:01:59.901861Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2601.06199","last_updated":"2026-06-01T03:39:22Z","snapshot_observed_at":"2026-08-06T01:59:30.134013Z","submitted_at":"2026-01-08T07:46:03Z","title":"FastSLM: Hierarchical Temporal Abstraction for Efficient Long-Form Speech Adaptation","version":3},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-03T12:01:59.901861Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2601.06199"},"observation_digest":"sha256:c010fc41c84e5ab3f17192d1b1b1596be47680035c0a70aa82d7aef6cc15352e","observation_id":"b8a97e48-dde2-4ad5-8b68-ff191afc5bd4","resolution":{"observed_at":"2026-08-03T12:01:59.901861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-07-13T16:50:28.996341Z","title":"Arushi Goel, Sreyan Ghosh, Jaehyeon Kim, Sonal Ku- mar, Zhifeng Kong, Sang gil Lee, Chao-Han Huck Yang, Ramani Duraiswami, Dinesh Manocha, Rafael Valle, and Bryan Catanzaro","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.27667","last_updated":"2026-05-28T11:00:35Z","snapshot_observed_at":"2026-08-02T13:16:24.345827Z","submitted_at":"2026-03-29T12:32:04Z","title":"EvA: An Evidence-First Audio Understanding Paradigm for LALMs","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-13T16:50:28.996341Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2603.27667"},"observation_digest":"sha256:56b18d7b769a25f904cfb644c55fe9fae3c000541c17ccb2a8568060ed9958b7","observation_id":"807caf8f-a394-43e0-8e96-8cd188daff0d","resolution":{"observed_at":"2026-07-13T16:50:28.996341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":"2406.11768","doi":"10.48550/arxiv.2406.11768","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GAMA: A large audio- language model with advanced audio understanding and complex rea- soning abilities","venue":"arXiv (Cornell University)","work_id":"f407a4af-a90d-4687-b15e-70573cc84a5d","year":2024},"citing_paper":{"arxiv_id":"2604.09721","last_updated":"2026-04-08T16:42:50Z","snapshot_observed_at":"2026-07-06T22:58:29.952947Z","submitted_at":"2026-04-08T16:42:50Z","title":"Jamendo-MT-QA: A Benchmark for Multi-Track Comparative Music Question Answering","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T17:16:59.034480Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2604.09721"},"observation_digest":"sha256:aabad71ac7dc07d3f01a645fc92cc5b6ca2b96f9d3dc4b472bdecb73a9b76588","observation_id":"2897d192-ccdc-40cd-ae15-6f9526583675","resolution":{"observed_at":"2026-05-11T07:10:58.086055Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":"2406.11768","doi":"10.48550/arxiv.2406.11768","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GAMA: A large audio- language model with advanced audio understanding and complex rea- soning abilities","venue":"arXiv (Cornell University)","work_id":"f407a4af-a90d-4687-b15e-70573cc84a5d","year":2024},"citing_paper":{"arxiv_id":"2604.14806","last_updated":"2026-04-16T09:30:13Z","snapshot_observed_at":"2026-08-01T04:51:16.142077Z","submitted_at":"2026-04-16T09:30:13Z","title":"Listen, Pause, and Reason: Toward Perception-Grounded Hybrid Reasoning for Audio Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T09:24:25.616750Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2604.14806"},"observation_digest":"sha256:575f119fdf8d6abb279624bef70f85d3ecdcfa2a9f2d7dbd4f28b88d367698fe","observation_id":"e8da2d69-0366-4eef-81b8-3841b3aeff5e","resolution":{"observed_at":"2026-05-10T09:28:38.897129Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":"2406.11768","doi":"10.48550/arxiv.2406.11768","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GAMA: A large audio- language model with advanced audio understanding and complex rea- soning abilities","venue":"arXiv (Cornell University)","work_id":"f407a4af-a90d-4687-b15e-70573cc84a5d","year":2024},"citing_paper":{"arxiv_id":"2604.15383","last_updated":"2026-04-16T02:30:41Z","snapshot_observed_at":"2026-07-06T23:02:58.361418Z","submitted_at":"2026-04-16T02:30:41Z","title":"Temporal Contrastive Decoding: A Training-Free Method for Large Audio-Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T10:16:26.526986Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2604.15383"},"observation_digest":"sha256:3e652500b488a82cb3d39bbed4e139714d1790486ab2e90d6fbc60a0e91c1bcc","observation_id":"d74d148d-6114-4231-a16b-f7a6508132cf","resolution":{"observed_at":"2026-05-10T10:19:20.685275Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":"2406.11768","doi":"10.48550/arxiv.2406.11768","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GAMA: A large audio- language model with advanced audio understanding and complex rea- soning abilities","venue":"arXiv (Cornell University)","work_id":"f407a4af-a90d-4687-b15e-70573cc84a5d","year":2024},"citing_paper":{"arxiv_id":"2605.23619","last_updated":"2026-05-22T13:26:47Z","snapshot_observed_at":"2026-07-06T23:33:49.013121Z","submitted_at":"2026-05-22T13:26:47Z","title":"Frame-Aligned Fusion of Canary and WavLM for Non-Intrusive Intelligibility Prediction of Hearing-Aid-Processed Speech","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-25T02:34:11.297579Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2605.23619"},"observation_digest":"sha256:182f77dca361a058c28f6abb56798e5384898deda8421b6a3de3f925ea3fdcba","observation_id":"34b70cd5-64b5-4786-8ee0-e3be2ff31c97","resolution":{"observed_at":"2026-05-25T02:35:14.575239Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-07-12T19:28:32.792370Z","title":"Gama: A large audio-language model with advanced audio understanding and complex reasoning abilities,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.05177","last_updated":"2026-04-17T12:31:17Z","snapshot_observed_at":"2026-08-02T21:40:51.010741Z","submitted_at":"2026-04-17T12:31:17Z","title":"MCBench: A Multicontext Safety Assessment Benchmark for Omni Large Language Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-07-12T19:28:32.792370Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2606.05177"},"observation_digest":"sha256:07a42a9a68bea217267dcd1b547284334b7ac8b3f2974bd75892ab1e6b29032d","observation_id":"6fed932d-bdcb-4364-b48d-b905fc9f6aa3","resolution":{"observed_at":"2026-07-12T19:28:32.792370Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":"2406.11768","doi":"10.48550/arxiv.2406.11768","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GAMA: A large audio- language model with advanced audio understanding and complex rea- soning abilities","venue":"arXiv (Cornell University)","work_id":"f407a4af-a90d-4687-b15e-70573cc84a5d","year":2024},"citing_paper":{"arxiv_id":"2606.08425","last_updated":"2026-06-20T19:16:20Z","snapshot_observed_at":"2026-08-06T01:18:08.801931Z","submitted_at":"2026-06-07T02:50:24Z","title":"TinyGiantALM: A Compact Audio-Language Model for Intent-Aware Reasoning under Resource Constraints","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-27T18:18:41.816342Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2606.08425"},"observation_digest":"sha256:934716656b5d0d3ff94f6cc35439495d91e432a5d8e43b5ec5bd6629c40ae8c7","observation_id":"b41efbd5-ea20-4409-a767-f73043369a62","resolution":{"observed_at":"2026-07-02T23:17:29.830453Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":"2406.11768","doi":"10.48550/arxiv.2406.11768","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GAMA: A large audio- language model with advanced audio understanding and complex rea- soning abilities","venue":"arXiv (Cornell University)","work_id":"f407a4af-a90d-4687-b15e-70573cc84a5d","year":2024},"citing_paper":{"arxiv_id":"2606.11219","last_updated":"2026-05-11T20:27:40Z","snapshot_observed_at":"2026-08-05T20:46:41.277498Z","submitted_at":"2026-05-11T20:27:40Z","title":"Afrispeech Semantics: Evaluating Audio Semantic Reasoning in Spoken Language Models Across Domains and Accents","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-06-30T22:11:44.891731Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2606.11219"},"observation_digest":"sha256:2f67d3aa2e8929e6e03f058e975fd0a0287aa012fb319389ff8f2c2ac55be05a","observation_id":"442e0517-a4c0-4866-b694-c9b6a67c0ec6","resolution":{"observed_at":"2026-06-30T22:15:05.181500Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":"2406.11768","doi":"10.48550/arxiv.2406.11768","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GAMA: A large audio- language model with advanced audio understanding and complex rea- soning abilities","venue":"arXiv (Cornell University)","work_id":"f407a4af-a90d-4687-b15e-70573cc84a5d","year":2024},"citing_paper":{"arxiv_id":"2606.23243","last_updated":"2026-06-22T12:28:33Z","snapshot_observed_at":"2026-08-03T08:35:01.520677Z","submitted_at":"2026-06-22T12:28:33Z","title":"Unlocking In-Context Learning in Audio-Language Models from Decentralized Medical Audio","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-26T09:16:25.450336Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2606.23243"},"observation_digest":"sha256:d7e338b5bbf7927719b3b7536cd14ee18f9dec2d05ce7dc995580dd487c4baff","observation_id":"4f5c37bc-1184-4884-bf90-ea0a7a740cfe","resolution":{"observed_at":"2026-07-04T09:59:45.495210Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11768","snapshot_observed_at":"2026-07-14T12:48:58.688011Z","title":"arXiv preprint arXiv:2406.11768 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.10299","last_updated":"2026-07-11T13:04:24Z","snapshot_observed_at":"2026-08-06T08:13:12.036559Z","submitted_at":"2026-07-11T13:04:24Z","title":"Empowering Long-form Omni-modal Understanding with Robust Audio Perception","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-14T12:48:58.688011Z"},"links":{"cited_paper":"/paper/2406.11768","citing_paper":"/paper/2607.10299"},"observation_digest":"sha256:8879d13ec95777e9bc6ca4c74eb75df51cae34be9c218a99fa75d2c80e264502","observation_id":"c8e42549-7300-4a39-b422-21309c1f4c80","resolution":{"observed_at":"2026-07-14T12:48:58.688011Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2406.11768/citation-record","integrity":"/paper/2406.11768/integrity","json":"/paper/2406.11768/citation-record.json","paper":"/paper/2406.11768"},"outbound":[],"paper":{"arxiv_id":"2406.11768","last_updated":"2024-06-17T17:31:01Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-01T20:00:05.737836Z","submitted_at":"2024-06-17T17:31:01Z","title":"GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 21 inbound Pith citation observations for arXiv:2406.11768."}