{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:J3AKKZS44BOJOXFNP3QGVNIWQB","short_pith_number":"pith:J3AKKZS4","schema_version":"1.0","canonical_sha256":"4ec0a5665ce05c975cad7ee06ab51680540b8d53fa9b4b46ecc89ce32b91749d","source":{"kind":"arxiv","id":"2404.00656","version":3},"attestation_state":"computed","paper":{"title":"WavLLM: Towards Robust and Adaptive Speech Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Furu Wei, Hongkun Hao, Jing Pan, Jinyu Li, Lingwei Meng, Linquan Liu, Long Zhou, Sanyuan Chen, Shujie Hu, Shujie Liu, Sunit Sivasankaran, Xunying Liu","submitted_at":"2024-03-31T12:01:32Z","abstract_excerpt":"The recent advancements in large language models (LLMs) have revolutionized the field of natural language processing, progressively broadening their scope to multimodal perception and generation. However, effectively integrating listening capabilities into LLMs poses significant challenges, particularly with respect to generalizing across varied contexts and executing complex auditory tasks. In this work, we introduce WavLLM, a robust and adaptive speech large language model with dual encoders, and a prompt-aware LoRA weight adapter, optimized by a two-stage curriculum learning approach. Lever"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.00656","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-31T12:01:32Z","cross_cats_sorted":["cs.AI","cs.SD","eess.AS"],"title_canon_sha256":"8c60a418c7558752ad5dbfc491a1dcbb60ac05b5d477a952fe316070629eee8c","abstract_canon_sha256":"669530498f19b5757c352e7796c9453556f27d265e204c78e73ec8f3075d518a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:09:58.113662Z","signature_b64":"KBf7XforakvIacLeX9Ugu0Ak6ofwXWUmVGstZ1nqbJdjDrMiL3XbJVaBMWenCeU6KVM1DWVlSaYd+ZqrIv0aAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4ec0a5665ce05c975cad7ee06ab51680540b8d53fa9b4b46ecc89ce32b91749d","last_reissued_at":"2026-07-05T09:09:58.113165Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:09:58.113165Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WavLLM: Towards Robust and Adaptive Speech Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Furu Wei, Hongkun Hao, Jing Pan, Jinyu Li, Lingwei Meng, Linquan Liu, Long Zhou, Sanyuan Chen, Shujie Hu, Shujie Liu, Sunit Sivasankaran, Xunying Liu","submitted_at":"2024-03-31T12:01:32Z","abstract_excerpt":"The recent advancements in large language models (LLMs) have revolutionized the field of natural language processing, progressively broadening their scope to multimodal perception and generation. However, effectively integrating listening capabilities into LLMs poses significant challenges, particularly with respect to generalizing across varied contexts and executing complex auditory tasks. In this work, we introduce WavLLM, a robust and adaptive speech large language model with dual encoders, and a prompt-aware LoRA weight adapter, optimized by a two-stage curriculum learning approach. Lever"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.00656","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.00656/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.00656","created_at":"2026-07-05T09:09:58.113228+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.00656v3","created_at":"2026-07-05T09:09:58.113228+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.00656","created_at":"2026-07-05T09:09:58.113228+00:00"},{"alias_kind":"pith_short_12","alias_value":"J3AKKZS44BOJ","created_at":"2026-07-05T09:09:58.113228+00:00"},{"alias_kind":"pith_short_16","alias_value":"J3AKKZS44BOJOXFN","created_at":"2026-07-05T09:09:58.113228+00:00"},{"alias_kind":"pith_short_8","alias_value":"J3AKKZS4","created_at":"2026-07-05T09:09:58.113228+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.06827","citing_title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05365","citing_title":"SPEARBench: A Benchmark for Naturalness Evaluation in Streaming Speech-to-Speech Language Models","ref_index":25,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26824","citing_title":"wav2tok 2.0: Scalable Audio Tokenization Maintaining Explicit Pairwise Token Alignment for Efficient Audio Retrieval","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01960","citing_title":"NAVER LABS Europe Submission to the Instruction-following 2026 Short Track","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12199","citing_title":"Which Speech Representation Better Matches Text-Native Reasoning? A Study of Speech-Text Alignment on Frame Rate and Representation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00247","citing_title":"Adaptive Perturbation Selection for Contrastive Audio Decoding","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2509.03526","citing_title":"Enhancing Speech Large Language Models through Reinforced Behavior Alignment","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2502.11946","citing_title":"Step-Audio: Unified Understanding and Generation in Intelligent Speech Interaction","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J3AKKZS44BOJOXFNP3QGVNIWQB","json":"https://pith.science/pith/J3AKKZS44BOJOXFNP3QGVNIWQB.json","graph_json":"https://pith.science/api/pith-number/J3AKKZS44BOJOXFNP3QGVNIWQB/graph.json","events_json":"https://pith.science/api/pith-number/J3AKKZS44BOJOXFNP3QGVNIWQB/events.json","paper":"https://pith.science/paper/J3AKKZS4"},"agent_actions":{"view_html":"https://pith.science/pith/J3AKKZS44BOJOXFNP3QGVNIWQB","download_json":"https://pith.science/pith/J3AKKZS44BOJOXFNP3QGVNIWQB.json","view_paper":"https://pith.science/paper/J3AKKZS4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.00656&json=true","fetch_graph":"https://pith.science/api/pith-number/J3AKKZS44BOJOXFNP3QGVNIWQB/graph.json","fetch_events":"https://pith.science/api/pith-number/J3AKKZS44BOJOXFNP3QGVNIWQB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J3AKKZS44BOJOXFNP3QGVNIWQB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J3AKKZS44BOJOXFNP3QGVNIWQB/action/storage_attestation","attest_author":"https://pith.science/pith/J3AKKZS44BOJOXFNP3QGVNIWQB/action/author_attestation","sign_citation":"https://pith.science/pith/J3AKKZS44BOJOXFNP3QGVNIWQB/action/citation_signature","submit_replication":"https://pith.science/pith/J3AKKZS44BOJOXFNP3QGVNIWQB/action/replication_record"}},"created_at":"2026-07-05T09:09:58.113228+00:00","updated_at":"2026-07-05T09:09:58.113228+00:00"}