{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XD7UAYH7DFAW3HVNAECDVBXP3K","short_pith_number":"pith:XD7UAYH7","schema_version":"1.0","canonical_sha256":"b8ff4060ff19416d9ead01043a86efdab4774bc3ba8dd49ce690e3397e4d6a34","source":{"kind":"arxiv","id":"2410.03751","version":4},"attestation_state":"computed","paper":{"title":"Recent Advances in Speech Language Models: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Dianzhi Yu, Guangyan Zhang, Irwin King, Qichao Wang, Wenqian Cui, Xiaoqi Jiao, Yiwen Guo, Ziqiao Meng","submitted_at":"2024-10-01T21:48:12Z","abstract_excerpt":"Large Language Models (LLMs) have recently garnered significant attention, primarily for their capabilities in text-based interactions. However, natural human interaction often relies on speech, necessitating a shift towards voice-based models. A straightforward approach to achieve this involves a pipeline of ``Automatic Speech Recognition (ASR) + LLM + Text-to-Speech (TTS)\", where input speech is transcribed to text, processed by an LLM, and then converted back to speech. Despite being straightforward, this method suffers from inherent limitations, such as information loss during modality con"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.03751","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-01T21:48:12Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"635db2684ef37c9a5d0bff65bf49a80b140b6fa5c73ad4bca220a95c74fa74d3","abstract_canon_sha256":"ebe475e50865e45a7f19f4be1a014a1b93090876272170fc558a452e01a2b5c3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:47.086223Z","signature_b64":"FqwhKHRfvyMUB3GnwMW3bcAPQiHVwiUJNsvu9EPxN5Gf4hOQTMPdQpUn+pY81L6gdxfDXjgi5dCpMwYLAF+XAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b8ff4060ff19416d9ead01043a86efdab4774bc3ba8dd49ce690e3397e4d6a34","last_reissued_at":"2026-07-05T11:49:47.085718Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:47.085718Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Recent Advances in Speech Language Models: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Dianzhi Yu, Guangyan Zhang, Irwin King, Qichao Wang, Wenqian Cui, Xiaoqi Jiao, Yiwen Guo, Ziqiao Meng","submitted_at":"2024-10-01T21:48:12Z","abstract_excerpt":"Large Language Models (LLMs) have recently garnered significant attention, primarily for their capabilities in text-based interactions. However, natural human interaction often relies on speech, necessitating a shift towards voice-based models. A straightforward approach to achieve this involves a pipeline of ``Automatic Speech Recognition (ASR) + LLM + Text-to-Speech (TTS)\", where input speech is transcribed to text, processed by an LLM, and then converted back to speech. Despite being straightforward, this method suffers from inherent limitations, such as information loss during modality con"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.03751","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.03751/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.03751","created_at":"2026-07-05T11:49:47.085785+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.03751v4","created_at":"2026-07-05T11:49:47.085785+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.03751","created_at":"2026-07-05T11:49:47.085785+00:00"},{"alias_kind":"pith_short_12","alias_value":"XD7UAYH7DFAW","created_at":"2026-07-05T11:49:47.085785+00:00"},{"alias_kind":"pith_short_16","alias_value":"XD7UAYH7DFAW3HVN","created_at":"2026-07-05T11:49:47.085785+00:00"},{"alias_kind":"pith_short_8","alias_value":"XD7UAYH7","created_at":"2026-07-05T11:49:47.085785+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.28480","citing_title":"Audio-Mind: An Auditable Agentic Framework for Audio Understanding","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2504.08528","citing_title":"On The Landscape of Spoken Language Models: A Comprehensive Survey","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2603.17837","citing_title":"The Silent Thought: Modeling Internal Cognition in Full-Duplex Spoken Dialogue Models via Latent Reasoning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2506.03610","citing_title":"Orak: A Foundational Benchmark for Training and Evaluating LLM Agents on Diverse Video Games","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08031","citing_title":"AU-Harness: An Open-Source Toolkit for Holistic Evaluation of Audio LLMs","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2510.06201","citing_title":"TokenChain: A Discrete Speech Chain via Semantic Token Modeling","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20657","citing_title":"Intelligent Agents with Emotional Intelligence: Current Trends, Challenges, and Future Prospects","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2510.09592","citing_title":"Mind-Paced Speaking: A Dual-Brain Approach to Real-Time Reasoning in Spoken Language Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2603.17837","citing_title":"The Silent Thought: Modeling Internal Cognition in Full-Duplex Spoken Dialogue Models via Latent Reasoning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13067","citing_title":"From Seeing it to Experiencing it: Interactive Evaluation of Intersectional Voice Bias in Human-AI Speech Interaction","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XD7UAYH7DFAW3HVNAECDVBXP3K","json":"https://pith.science/pith/XD7UAYH7DFAW3HVNAECDVBXP3K.json","graph_json":"https://pith.science/api/pith-number/XD7UAYH7DFAW3HVNAECDVBXP3K/graph.json","events_json":"https://pith.science/api/pith-number/XD7UAYH7DFAW3HVNAECDVBXP3K/events.json","paper":"https://pith.science/paper/XD7UAYH7"},"agent_actions":{"view_html":"https://pith.science/pith/XD7UAYH7DFAW3HVNAECDVBXP3K","download_json":"https://pith.science/pith/XD7UAYH7DFAW3HVNAECDVBXP3K.json","view_paper":"https://pith.science/paper/XD7UAYH7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.03751&json=true","fetch_graph":"https://pith.science/api/pith-number/XD7UAYH7DFAW3HVNAECDVBXP3K/graph.json","fetch_events":"https://pith.science/api/pith-number/XD7UAYH7DFAW3HVNAECDVBXP3K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XD7UAYH7DFAW3HVNAECDVBXP3K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XD7UAYH7DFAW3HVNAECDVBXP3K/action/storage_attestation","attest_author":"https://pith.science/pith/XD7UAYH7DFAW3HVNAECDVBXP3K/action/author_attestation","sign_citation":"https://pith.science/pith/XD7UAYH7DFAW3HVNAECDVBXP3K/action/citation_signature","submit_replication":"https://pith.science/pith/XD7UAYH7DFAW3HVNAECDVBXP3K/action/replication_record"}},"created_at":"2026-07-05T11:49:47.085785+00:00","updated_at":"2026-07-05T11:49:47.085785+00:00"}