{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZBN5E556KAVAJC35N4NS7M25DS","short_pith_number":"pith:ZBN5E556","schema_version":"1.0","canonical_sha256":"c85bd277be502a048b7d6f1b2fb35d1ca9d17f1aba0f8897b31ecfd8b1b79651","source":{"kind":"arxiv","id":"2402.05706","version":3},"attestation_state":"computed","paper":{"title":"Paralinguistics-Aware Speech-Empowered Large Language Models for Natural Conversation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Eunwoo Song, Heeseung Kim, Jaehong Lee, Jungwhan Kim, Jung-Woo Ha, Kang Min Yoo, Kyeongseok Jeong, Myungwoo Oh, Ohsung Kwon, Soonshin Seo, Soyoon Kim, Sungroh Yoon","submitted_at":"2024-02-08T14:35:09Z","abstract_excerpt":"Recent work shows promising results in expanding the capabilities of large language models (LLM) to directly understand and synthesize speech. However, an LLM-based strategy for modeling spoken dialogs remains elusive, calling for further investigation. This paper introduces an extensive speech-text LLM framework, the Unified Spoken Dialog Model (USDM), designed to generate coherent spoken responses with naturally occurring prosodic features relevant to the given input speech without relying on explicit automatic speech recognition (ASR) or text-to-speech (TTS) systems. We have verified the in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.05706","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-08T14:35:09Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"7b37b2fb05bf1d9435cf3a102f80298b32e9fc81b555b39d4042f474d8223df3","abstract_canon_sha256":"db6f257ba52e46a666a06303949417ef26f76816d8d00a0f094cd7785213155d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:31.838684Z","signature_b64":"zmt5t1XNjlGtvBbJpB5HO+pg6vcMkquf92A2SQMoMajAbnLK8abcTXsMyflujLoE5Ka51YDtEX3uKG8pO05nAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c85bd277be502a048b7d6f1b2fb35d1ca9d17f1aba0f8897b31ecfd8b1b79651","last_reissued_at":"2026-07-05T09:41:31.838163Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:31.838163Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Paralinguistics-Aware Speech-Empowered Large Language Models for Natural Conversation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Eunwoo Song, Heeseung Kim, Jaehong Lee, Jungwhan Kim, Jung-Woo Ha, Kang Min Yoo, Kyeongseok Jeong, Myungwoo Oh, Ohsung Kwon, Soonshin Seo, Soyoon Kim, Sungroh Yoon","submitted_at":"2024-02-08T14:35:09Z","abstract_excerpt":"Recent work shows promising results in expanding the capabilities of large language models (LLM) to directly understand and synthesize speech. However, an LLM-based strategy for modeling spoken dialogs remains elusive, calling for further investigation. This paper introduces an extensive speech-text LLM framework, the Unified Spoken Dialog Model (USDM), designed to generate coherent spoken responses with naturally occurring prosodic features relevant to the given input speech without relying on explicit automatic speech recognition (ASR) or text-to-speech (TTS) systems. We have verified the in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.05706","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.05706/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.05706","created_at":"2026-07-05T09:41:31.838230+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.05706v3","created_at":"2026-07-05T09:41:31.838230+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.05706","created_at":"2026-07-05T09:41:31.838230+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZBN5E556KAVA","created_at":"2026-07-05T09:41:31.838230+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZBN5E556KAVAJC35","created_at":"2026-07-05T09:41:31.838230+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZBN5E556","created_at":"2026-07-05T09:41:31.838230+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.03526","citing_title":"Enhancing Speech Large Language Models through Reinforced Behavior Alignment","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZBN5E556KAVAJC35N4NS7M25DS","json":"https://pith.science/pith/ZBN5E556KAVAJC35N4NS7M25DS.json","graph_json":"https://pith.science/api/pith-number/ZBN5E556KAVAJC35N4NS7M25DS/graph.json","events_json":"https://pith.science/api/pith-number/ZBN5E556KAVAJC35N4NS7M25DS/events.json","paper":"https://pith.science/paper/ZBN5E556"},"agent_actions":{"view_html":"https://pith.science/pith/ZBN5E556KAVAJC35N4NS7M25DS","download_json":"https://pith.science/pith/ZBN5E556KAVAJC35N4NS7M25DS.json","view_paper":"https://pith.science/paper/ZBN5E556","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.05706&json=true","fetch_graph":"https://pith.science/api/pith-number/ZBN5E556KAVAJC35N4NS7M25DS/graph.json","fetch_events":"https://pith.science/api/pith-number/ZBN5E556KAVAJC35N4NS7M25DS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZBN5E556KAVAJC35N4NS7M25DS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZBN5E556KAVAJC35N4NS7M25DS/action/storage_attestation","attest_author":"https://pith.science/pith/ZBN5E556KAVAJC35N4NS7M25DS/action/author_attestation","sign_citation":"https://pith.science/pith/ZBN5E556KAVAJC35N4NS7M25DS/action/citation_signature","submit_replication":"https://pith.science/pith/ZBN5E556KAVAJC35N4NS7M25DS/action/replication_record"}},"created_at":"2026-07-05T09:41:31.838230+00:00","updated_at":"2026-07-05T09:41:31.838230+00:00"}