{"as_of":"2026-08-07T22:05:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:cc389b5956dc58b47ae31c341bcb0766b9d1a0aa47a84c6b1566213707e4a0b7","coverage":[{"denominator":34,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":34,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T21:55:55.302080Z","state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-21T02:08:06.976461Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-21T02:09:24.310163Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"cited_work":{"arxiv_id":"2506.23049","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.23049","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Aura: Agent for understanding, reasoning, and auto- mated tool use in voice-driven tasks","venue":null,"work_id":"08ca7547-c4d0-48a5-b956-cf3060d2627d","year":2025},"citing_paper":{"arxiv_id":"2604.15037","last_updated":"2026-05-02T12:33:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-16T14:06:30Z","title":"From Reactive to Proactive: Assessing the Proactivity of Voice Agents via ProVoice-Bench","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T11:44:12.373082Z"},"links":{"cited_paper":"/paper/2506.23049","citing_paper":"/paper/2604.15037"},"observation_digest":"sha256:b71e58cbacc959ec319416f1ca24f118e527693605f557905827b7d0bf273044","observation_id":"9231afea-d3b6-4e49-9b70-35d3f6ae2021","resolution":{"observed_at":"2026-05-10T11:45:21.002818Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"cited_work":{"arxiv_id":"2506.23049","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.23049","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Aura: Agent for understanding, reasoning, and auto- mated tool use in voice-driven tasks","venue":null,"work_id":"08ca7547-c4d0-48a5-b956-cf3060d2627d","year":2025},"citing_paper":{"arxiv_id":"2605.21008","last_updated":"2026-05-20T10:44:56Z","snapshot_observed_at":"2026-07-06T23:31:32.577242Z","submitted_at":"2026-05-20T10:44:56Z","title":"A Survey of Audio Reasoning in Multimodal Foundation Models","version":1},"reference_index":109,"source":"pdf_text","source_observed_at":"2026-05-21T02:08:06.976461Z"},"links":{"cited_paper":"/paper/2506.23049","citing_paper":"/paper/2605.21008"},"observation_digest":"sha256:2ef69854127f8c08f423ab5f03aa68d7412b50172470b0a3dc4cd7049ce4bb98","observation_id":"878f8158-8b47-4503-8238-062795f7805e","resolution":{"observed_at":"2026-05-21T02:09:24.313261Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.23049/citation-record","integrity":"/paper/2506.23049/integrity","json":"/paper/2506.23049/citation-record.json","paper":"/paper/2506.23049"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:57.292892Z","title":"Espnet-sds: Unified toolkit and demo for spoken dialogue systems,","venue":null,"work_id":"fe156377-f1a5-451b-9309-de0e20fb9987","year":null},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.060791Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:268a1780ead9d1250f08fd5c69543986454e9474db60e926de5b5f00939352a2","observation_id":"6a050def-1168-4fe9-8295-ecffce12c92a","resolution":{"observed_at":"2026-08-06T21:55:57.345167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11000","last_updated":"2023-05-19T14:41:16Z","snapshot_observed_at":"2026-08-07T10:56:05.622094Z","submitted_at":"2023-05-18T14:23:25Z","title":"SpeechGPT: Empowering Large Language Models with Intrinsic Cross-Modal Conversational Abilities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11000","snapshot_observed_at":"2026-08-06T21:55:53.134838Z","title":"Speechgpt: Empowering large language models with intrinsic cross-modal conversational abilities,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.134838Z"},"links":{"cited_paper":"/paper/2305.11000","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:e937dc3867be56597112115659b5a4a7b18f2ff3d3d4d20501db9403cfaac106","observation_id":"0dc7d8ef-a7bb-4d91-b572-23ebe7d1ba76","resolution":{"observed_at":"2026-08-06T21:55:53.134838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17239","last_updated":"2025-02-24T15:16:34Z","snapshot_observed_at":"2026-08-07T17:53:08.473463Z","submitted_at":"2025-02-24T15:16:34Z","title":"Baichuan-Audio: A Unified Framework for End-to-End Speech Interaction","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.17239","snapshot_observed_at":"2026-08-06T21:55:53.227261Z","title":"Baichuan- audio: A unified framework for end-to-end speech interaction,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.227261Z"},"links":{"cited_paper":"/paper/2502.17239","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:5b49c2a67c8d9dba6608c554a41d62e7904d9526c9f4534f328e28f84f35ba5d","observation_id":"be26c935-a7ac-46cc-8640-094e2ff6b65b","resolution":{"observed_at":"2026-08-06T21:55:53.227261Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.16725","last_updated":"2024-11-05T02:24:18Z","snapshot_observed_at":"2026-07-06T19:07:46.545514Z","submitted_at":"2024-08-29T17:18:53Z","title":"Mini-Omni: Language Models Can Hear, Talk While Thinking in Streaming","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.16725","snapshot_observed_at":"2026-08-06T21:55:53.426594Z","title":"Mini-omni: Language models can hear, talk while thinking in streaming,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.426594Z"},"links":{"cited_paper":"/paper/2408.16725","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:5fc3c10e970c09f977dbdd56a3b26f89ee72100c28826d601f6cf1f247f20fb2","observation_id":"83cfe486-8963-4984-b578-6021b9fedb3c","resolution":{"observed_at":"2026-08-06T21:55:53.426594Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:53.535944Z","title":"Openomni: Advancing open-source omnimodal large language models with progressive multimodal alignment and real-time self-aware emotional speech synthesis,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.535944Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:32d2b0139248ee2b6593766885d207874065af71cadb74cecd7f40fa609a0b5b","observation_id":"6c4f318d-5e12-4868-9bfc-5745befd587c","resolution":{"observed_at":"2026-08-06T21:55:53.535944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17196","last_updated":"2024-12-11T15:45:21Z","snapshot_observed_at":"2026-07-06T19:37:56.214143Z","submitted_at":"2024-10-22T17:15:20Z","title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17196","snapshot_observed_at":"2026-08-06T21:55:53.632465Z","title":"V oicebench: Benchmarking llm-based voice assistants,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.632465Z"},"links":{"cited_paper":"/paper/2410.17196","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:9f87955d254ea043de2b5a6022776e4608815d006a6e99b734b0fb18bc56f574","observation_id":"29136298-6744-4ed0-ad02-6f8d95d881cf","resolution":{"observed_at":"2026-08-06T21:55:53.632465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1810.00278","last_updated":"2020-04-20T15:02:43Z","snapshot_observed_at":"2026-08-02T23:08:28.054429Z","submitted_at":"2018-09-29T23:44:39Z","title":"MultiWOZ -- A Large-Scale Multi-Domain Wizard-of-Oz Dataset for Task-Oriented Dialogue Modelling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.00278","snapshot_observed_at":"2026-08-06T21:55:53.745815Z","title":"Multiwoz–a large-scale multi-domain wizard- of-oz dataset for task-oriented dialogue modelling,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.745815Z"},"links":{"cited_paper":"/paper/1810.00278","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:dbd85e932a61b2098b15547358d16adc7b77e82275242af78510839545a91057","observation_id":"0a0bb2e1-e7cd-43f1-8433-9cbfe2ada353","resolution":{"observed_at":"2026-08-06T21:55:53.745815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13040","last_updated":"2025-06-24T07:06:57Z","snapshot_observed_at":"2026-07-06T15:30:42.173342Z","submitted_at":"2023-05-22T13:47:51Z","title":"SpokenWOZ: A Large-Scale Speech-Text Benchmark for Spoken Task-Oriented Dialogue Agents","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13040","snapshot_observed_at":"2026-08-06T21:55:53.858082Z","title":"Spokenwoz: A large-scale speech-text benchmark for spoken task-oriented dialogue agents,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.858082Z"},"links":{"cited_paper":"/paper/2305.13040","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:48e1532a51053c9f1bf9b9f18b6d7ef3ddf067f62a9783315587f1ee1aa30b52","observation_id":"1729b4ea-7e13-4fd2-8e35-6e6537c543bc","resolution":{"observed_at":"2026-08-06T21:55:53.858082Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02155","last_updated":"2022-03-04T07:04:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-04T07:04:42Z","title":"Training language models to follow instructions with human feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.02155","snapshot_observed_at":"2026-08-06T21:55:53.956373Z","title":"Training language models to follow instructions with human feedback,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.956373Z"},"links":{"cited_paper":"/paper/2203.02155","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:afd9f07718d10e74347889ad44942cc1aabfb7fce124d5a52655bfef24e018ca","observation_id":"1b3c17e9-60ec-4a67-93ca-76db04cd35ad","resolution":{"observed_at":"2026-08-06T21:55:53.956373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.17580","last_updated":"2023-12-03T18:17:21Z","snapshot_observed_at":"2026-08-03T00:52:54.308486Z","submitted_at":"2023-03-30T17:48:28Z","title":"HuggingGPT: Solving AI Tasks with ChatGPT and its Friends in Hugging Face","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.17580","snapshot_observed_at":"2026-08-06T21:55:54.050660Z","title":"Hugginggpt: Solving ai tasks with chatgpt and its friends in hugging face,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.050660Z"},"links":{"cited_paper":"/paper/2303.17580","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:86ff52ff1caadf5448a8fd2337b217c7f2a5879805354c82ce3d7a42cca60ef7","observation_id":"556260fb-0aa7-484a-98ae-0209d064c99e","resolution":{"observed_at":"2026-08-06T21:55:54.050660Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-06T21:55:54.130776Z","title":"Api-bank: A comprehensive benchmark for tool-augmented llms,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.130776Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:fab72a70382374bcff430c0fd6ba6d434e39978ddf1bc87e6f90b8d9867fc1c7","observation_id":"5e8bd2b4-b095-4eaa-96f2-ef35ab2f289f","resolution":{"observed_at":"2026-08-06T21:55:54.130776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.00564","last_updated":"2025-03-01T17:23:51Z","snapshot_observed_at":"2026-08-07T17:37:04.045007Z","submitted_at":"2025-03-01T17:23:51Z","title":"ToolDial: Multi-turn Dialogue Generation Method for Tool-Augmented Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.00564","snapshot_observed_at":"2026-08-06T21:55:54.171642Z","title":"Tooldial: Multi-turn dialogue generation method for tool-augmented language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.171642Z"},"links":{"cited_paper":"/paper/2503.00564","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:23a10d4df4b2be85f69af943aa7c41b7c71b06a5b30a6aa157938e514cb4f45f","observation_id":"44b5b849-70de-40c6-b1f0-aaa0cafb8fa0","resolution":{"observed_at":"2026-08-06T21:55:54.171642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:57.197559Z","title":"Rethinking task-oriented dialogue systems: From complex modularity to zero-shot autonomous agent,","venue":null,"work_id":"c1a902f3-f777-4618-a4b3-727cccdf3660","year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.215277Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:e62bc7f849cf13476626fece2aaa8d5ac44d5a1c4c31fbb22cba3e6180a08ddf","observation_id":"6fb7d218-242a-4f0f-af21-6204b876b14c","resolution":{"observed_at":"2026-08-06T21:55:57.237848Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:54.285113Z","title":"React: Synergizing reasoning and acting in language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.285113Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:cce90e8925211ea5c5603dd0551e2fad15e726750913b60e0845e3d9dd75d0a3","observation_id":"a9230911-5ef5-4918-bd37-7749709b1904","resolution":{"observed_at":"2026-08-06T21:55:54.285113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:57.098888Z","title":"Audio-cot: Exploring chain-of-thought reasoning in large audio language model,","venue":null,"work_id":"be34954d-1e08-4534-8a9b-e65237e3e07f","year":null},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.342505Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:dafc6874ad46305afa5d5ba71e3a0d7125f9f13fb2771c559f6a48129263af8d","observation_id":"899075e1-9ab3-44f9-bedd-970ffb07a5f2","resolution":{"observed_at":"2026-08-06T21:55:57.139353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:54.436315Z","title":"Audio-reasoner: Improving reasoning capability in large audio language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.436315Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:2fdfc3a43659d2b92e6e28c81ae4042c004d18c07bb70df9fd5eb61dc95ee20c","observation_id":"75024118-a0c6-4ac4-b6bf-d45f86f1d41e","resolution":{"observed_at":"2026-08-06T21:55:54.436315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07246","last_updated":"2025-01-13T11:54:40Z","snapshot_observed_at":"2026-08-07T04:02:14.954506Z","submitted_at":"2025-01-13T11:54:40Z","title":"Audio-CoT: Exploring Chain-of-Thought Reasoning in Large Audio Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.07246","snapshot_observed_at":"2026-08-06T21:55:54.386404Z","title":"Available: https://arxiv.org/abs/2501.07246","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.386404Z"},"links":{"cited_paper":"/paper/2501.07246","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:cef3ff7550be85b176129683a4edf0abc2ed97a2c33e907cc7c580770fcad75d","observation_id":"13a23776-4513-41db-ab68-e23694a70cd5","resolution":{"observed_at":"2026-08-06T21:55:54.386404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:54.539532Z","title":"Can a suit of armor conduct electricity? a new dataset for open book question answering,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.539532Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:371ad5a2e27a0d01567605450c8c4aefccbff39ed079c14428ef2dbde391e265","observation_id":"5f6776fd-53f7-48e7-a592-eb9359226dfc","resolution":{"observed_at":"2026-08-06T21:55:54.539532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00927","last_updated":"2025-04-19T15:43:36Z","snapshot_observed_at":"2026-08-04T03:00:55.235335Z","submitted_at":"2024-11-01T15:57:45Z","title":"ReSpAct: Harmonizing Reasoning, Speaking, and Acting Towards Building Large Language Model-Based Conversational AI Agents","version":2},"cited_work":{"arxiv_id":"2411.00927","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.00927","snapshot_observed_at":"2026-08-06T21:55:55.409227Z","title":"ReSpAct: Harmonizing Reasoning, Speaking, and Acting Towards Building Large Language Model-Based Conversational AI Agents","venue":"cs.CL","work_id":"e7a3a49d-06db-411d-a014-44b3425ab948","year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.473272Z"},"links":{"cited_paper":"/paper/2411.00927","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:7be52dbc44cb134bb202af3c5e82f18d7c1a5ff66ade11a93bccd4e7a9878525","observation_id":"e08f5fdd-a684-4b24-9a06-34a544179e27","resolution":{"observed_at":"2026-08-06T21:55:55.474892Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:56.986840Z","title":"Gpt-4o: Openai’s new multimodal flagship model,","venue":null,"work_id":"ab951c5c-196c-47c3-bf34-8d3c017c42a9","year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.625379Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:f5bf435186345bb822b725fac2a235716654a016228bd81717e89dcd90bc25af","observation_id":"11404fd2-5124-4b62-a298-96fbd8fde36c","resolution":{"observed_at":"2026-08-06T21:55:57.037215Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.04356","last_updated":"2022-12-06T18:46:04Z","snapshot_observed_at":"2026-07-06T14:28:21.844826Z","submitted_at":"2022-12-06T18:46:04Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.04356","snapshot_observed_at":"2026-08-06T21:55:54.575759Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.575759Z"},"links":{"cited_paper":"/paper/2212.04356","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:945c3c6183198d45ec2d6c0ca3ab6162a3df61b1bc29ec16911668f655f4df05","observation_id":"13a632cc-e96a-46d8-820f-e45d952320fc","resolution":{"observed_at":"2026-08-06T21:55:54.575759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:56.687099Z","title":"Espnet-TTS: Unified, reproducible, and integratable open source end-to-end text-to-speech toolkit,","venue":null,"work_id":"d1c4bf4a-1b00-4022-b81a-69ea94421bfb","year":2020},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.714602Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:f66a029d59c2417329381eacc6166aa1b3eef87fd9c183b59884ddeb4d9c750a","observation_id":"3705390a-447d-4e6f-b7d1-6b65e8ad02ce","resolution":{"observed_at":"2026-08-06T21:55:56.795127Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:56.859364Z","title":"Owsm v3.1: Better and faster open whisper-style speech models based on e-branchformer,","venue":null,"work_id":"4f66d0ea-16f0-430c-9db2-a78208c6e42b","year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.650824Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:86e832ec293d780b2b873971dcee3718ceb55321d9892c5290222a40be9336ef","observation_id":"37471f67-73b0-461a-8515-e273a18636a3","resolution":{"observed_at":"2026-08-06T21:55:56.937871Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06180","last_updated":"2023-09-12T12:50:04Z","snapshot_observed_at":"2026-08-02T09:51:08.145755Z","submitted_at":"2023-09-12T12:50:04Z","title":"Efficient Memory Management for Large Language Model Serving with PagedAttention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06180","snapshot_observed_at":"2026-08-06T21:55:54.808526Z","title":"Efficient memory management for large language model serving with pagedattention,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.808526Z"},"links":{"cited_paper":"/paper/2309.06180","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:1a756c3e8f4ae20ce421c1970b6db0664edd794d5f5781069e20de2ecb4e285b","observation_id":"66ba6451-65ec-41fd-9e2c-1ef0c70223ec","resolution":{"observed_at":"2026-08-06T21:55:54.808526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T21:55:54.758002Z","title":"The llama 3 herd of models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.758002Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:bee46a1ca4eb81e41564e12ba8e1821d9d521ee03447a46df3c61fb6c1fbe960","observation_id":"4d4e4b70-79e5-4f2e-bca8-d5e1e66c4115","resolution":{"observed_at":"2026-08-06T21:55:54.758002Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:56.327294Z","title":"Gpt-4o: Openai’s new multimodal flagship model,","venue":null,"work_id":"289e120d-db94-4e5c-9587-4e21da7e2fc0","year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.920074Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:d5c5758eb88290e5b2ad3ff3bfd3da0c49c15b9392479f71448e18333e984cd8","observation_id":"c741c3e5-60be-4f0c-a95b-f7eb3d2c3196","resolution":{"observed_at":"2026-08-06T21:55:56.399442Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:56.474970Z","title":"Alpacaeval: An automatic evaluator of instruction-following models,","venue":null,"work_id":"94e4882c-6b69-4577-813b-8c3160cacae5","year":2023},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.850695Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:eed3ad1e7f9c7f7cc60f70251ab64054eb65b5de32e6d6fdee73a35f32f86c6e","observation_id":"64c3b58c-b427-4d79-946b-af719df12293","resolution":{"observed_at":"2026-08-06T21:55:56.544396Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00037","last_updated":"2024-10-02T09:11:45Z","snapshot_observed_at":"2026-07-30T10:21:14.474746Z","submitted_at":"2024-09-17T17:55:39Z","title":"Moshi: a speech-text foundation model for real-time dialogue","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00037","snapshot_observed_at":"2026-08-06T21:55:55.026534Z","title":"Moshi: a speech-text foundation model for real-time dialogue,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:55.026534Z"},"links":{"cited_paper":"/paper/2410.00037","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:557b9f0b7cc1df42ba8bad62c770d6a010cea6b42af5a916222665760a0f3b6f","observation_id":"4c43f8dc-f9de-4f62-8a4b-20a10ca0f237","resolution":{"observed_at":"2026-08-06T21:55:55.026534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:56.121062Z","title":"Gpt-4o: Openai’s new multimodal flagship model,","venue":null,"work_id":"e972c10a-c454-4d73-8882-5d971d9fa652","year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.953118Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:a953179689da7ea16cae1c301c075f168ee37a9677a10924aa9592f080ef0f5b","observation_id":"32a86f2f-3495-438e-ad5a-6029d8f42bd6","resolution":{"observed_at":"2026-08-06T21:55:56.224799Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.18425","last_updated":"2025-04-25T15:31:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-25T15:31:46Z","title":"Kimi-Audio Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.18425","snapshot_observed_at":"2026-08-06T21:55:55.166303Z","title":"Kimi-audio technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:55.166303Z"},"links":{"cited_paper":"/paper/2504.18425","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:f6499ec8ab8be0b5353e45cc9a553e5106ed6ab6b02ab41453fa82759f9865c3","observation_id":"f6d58d67-3f07-4053-a581-dfd13108f015","resolution":{"observed_at":"2026-08-06T21:55:55.166303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11190","last_updated":"2024-11-05T02:27:57Z","snapshot_observed_at":"2026-07-06T19:33:34.819837Z","submitted_at":"2024-10-15T02:10:45Z","title":"Mini-Omni2: Towards Open-source GPT-4o with Vision, Speech and Duplex Capabilities","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11190","snapshot_observed_at":"2026-08-06T21:55:55.082723Z","title":"Mini-omni2: Towards open-source gpt-4o with vision, speech and duplex capabilities,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:55.082723Z"},"links":{"cited_paper":"/paper/2410.11190","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:9be6fe9a0e1096b9a11d6abbbbebea273de3a69f8f938fd3c44a96321221d0e0","observation_id":"63d253aa-d644-45da-a7b8-a64c00d06d94","resolution":{"observed_at":"2026-08-06T21:55:55.082723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:55.925500Z","title":"Parakeet-tdt-0.6b-v2,","venue":null,"work_id":"ddfbbe92-1b75-4615-8d55-4fff2f374293","year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:55.302080Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:999526de59f5fc30bf7488485cbb6fded042fab77fba8068f03d8242864fc135","observation_id":"342d84ad-921d-4ceb-a81f-bd574f5321f8","resolution":{"observed_at":"2026-08-06T21:55:56.017766Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-06T21:55:55.221863Z","title":"Qwen3 technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:55.221863Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:6cf462bbdd627a29a95dd1c8a1d27df5ab1e618b0b33b1c0374cb762bf7f3bb0","observation_id":"7640f52d-8697-41b2-b649-200ececb7083","resolution":{"observed_at":"2026-08-06T21:55:55.221863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.08533","last_updated":"2025-03-11T15:24:02Z","snapshot_observed_at":"2026-08-07T17:12:39.670272Z","submitted_at":"2025-03-11T15:24:02Z","title":"ESPnet-SDS: Unified Toolkit and Demo for Spoken Dialogue Systems","version":1},"cited_work":{"arxiv_id":"2503.08533","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.08533","snapshot_observed_at":"2026-08-06T21:55:55.772229Z","title":"ESPnet-SDS: Unified Toolkit and Demo for Spoken Dialogue Systems","venue":"cs.CL","work_id":"abdcf874-c917-4d86-b45f-1cf470770a44","year":2025},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.100157Z"},"links":{"cited_paper":"/paper/2503.08533","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:5b17a23f8eb40efff6adc4bbdbde42d45e837d5fda4cc6bc016fc88011d7234a","observation_id":"853e5859-4f46-4486-a17e-4cc553b47172","resolution":{"observed_at":"2026-08-06T21:55:55.822601Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks"},"reference_resolution":{"displayed":34,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":22,"verified_exact":1,"verified_fuzzy":10},"total_outbound_references":34},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 34 of 34 outbound references and 2 inbound Pith citation observations for arXiv:2506.23049."}