{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JCLFS6KUWDJZ7DVU527XWPURAK","short_pith_number":"pith:JCLFS6KU","schema_version":"1.0","canonical_sha256":"4896597954b0d39f8eb4eebf7b3e9102954d28e9efcacdb38177434275565f26","source":{"kind":"arxiv","id":"2502.14727","version":1},"attestation_state":"computed","paper":{"title":"WavRAG: Audio-Integrated Retrieval Augmented Generation for Spoken Dialogue Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Haoxiao Wang, Jin Xu, Jinzheng He, Shengpeng Ji, Siyu Chen, Yifu Chen, Zhou Zhao, Ziqing Wang","submitted_at":"2025-02-20T16:54:07Z","abstract_excerpt":"Retrieval Augmented Generation (RAG) has gained widespread adoption owing to its capacity to empower large language models (LLMs) to integrate external knowledge. However, existing RAG frameworks are primarily designed for text-based LLMs and rely on Automatic Speech Recognition to process speech input, which discards crucial audio information, risks transcription errors, and increases computational overhead. Therefore, we introduce WavRAG, the first retrieval augmented generation framework with native, end-to-end audio support. WavRAG offers two key features: 1) Bypassing ASR, WavRAG directly"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.14727","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2025-02-20T16:54:07Z","cross_cats_sorted":["cs.AI","eess.AS"],"title_canon_sha256":"529e90a755d50a7f274015a66528a656b4d94c20c9a9cd3ce7d2aa2afe3dbfd3","abstract_canon_sha256":"5157b82dc1b87153f4ea319dd742c34661a0d74f9c47c3205ed28205fea2ad19"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:31.646350Z","signature_b64":"1WV8oupjqKI4b/G4J+axPkxTvILisiO4DvB7c/uYFQu0ecvBF03De6QjrACaTQ6MfbQ9JB/m6faUwKl7HbYrDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4896597954b0d39f8eb4eebf7b3e9102954d28e9efcacdb38177434275565f26","last_reissued_at":"2026-07-05T10:17:31.645869Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:31.645869Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WavRAG: Audio-Integrated Retrieval Augmented Generation for Spoken Dialogue Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Haoxiao Wang, Jin Xu, Jinzheng He, Shengpeng Ji, Siyu Chen, Yifu Chen, Zhou Zhao, Ziqing Wang","submitted_at":"2025-02-20T16:54:07Z","abstract_excerpt":"Retrieval Augmented Generation (RAG) has gained widespread adoption owing to its capacity to empower large language models (LLMs) to integrate external knowledge. However, existing RAG frameworks are primarily designed for text-based LLMs and rely on Automatic Speech Recognition to process speech input, which discards crucial audio information, risks transcription errors, and increases computational overhead. Therefore, we introduce WavRAG, the first retrieval augmented generation framework with native, end-to-end audio support. WavRAG offers two key features: 1) Bypassing ASR, WavRAG directly"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.14727","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.14727/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.14727","created_at":"2026-07-05T10:17:31.645924+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.14727v1","created_at":"2026-07-05T10:17:31.645924+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.14727","created_at":"2026-07-05T10:17:31.645924+00:00"},{"alias_kind":"pith_short_12","alias_value":"JCLFS6KUWDJZ","created_at":"2026-07-05T10:17:31.645924+00:00"},{"alias_kind":"pith_short_16","alias_value":"JCLFS6KUWDJZ7DVU","created_at":"2026-07-05T10:17:31.645924+00:00"},{"alias_kind":"pith_short_8","alias_value":"JCLFS6KU","created_at":"2026-07-05T10:17:31.645924+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.20267","citing_title":"ATIR: Towards Audio-Text Interleaved Contextual Retrieval","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JCLFS6KUWDJZ7DVU527XWPURAK","json":"https://pith.science/pith/JCLFS6KUWDJZ7DVU527XWPURAK.json","graph_json":"https://pith.science/api/pith-number/JCLFS6KUWDJZ7DVU527XWPURAK/graph.json","events_json":"https://pith.science/api/pith-number/JCLFS6KUWDJZ7DVU527XWPURAK/events.json","paper":"https://pith.science/paper/JCLFS6KU"},"agent_actions":{"view_html":"https://pith.science/pith/JCLFS6KUWDJZ7DVU527XWPURAK","download_json":"https://pith.science/pith/JCLFS6KUWDJZ7DVU527XWPURAK.json","view_paper":"https://pith.science/paper/JCLFS6KU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.14727&json=true","fetch_graph":"https://pith.science/api/pith-number/JCLFS6KUWDJZ7DVU527XWPURAK/graph.json","fetch_events":"https://pith.science/api/pith-number/JCLFS6KUWDJZ7DVU527XWPURAK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JCLFS6KUWDJZ7DVU527XWPURAK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JCLFS6KUWDJZ7DVU527XWPURAK/action/storage_attestation","attest_author":"https://pith.science/pith/JCLFS6KUWDJZ7DVU527XWPURAK/action/author_attestation","sign_citation":"https://pith.science/pith/JCLFS6KUWDJZ7DVU527XWPURAK/action/citation_signature","submit_replication":"https://pith.science/pith/JCLFS6KUWDJZ7DVU527XWPURAK/action/replication_record"}},"created_at":"2026-07-05T10:17:31.645924+00:00","updated_at":"2026-07-05T10:17:31.645924+00:00"}