{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4G3FMJ5U5DOFG3SO7YQLCUH7TN","short_pith_number":"pith:4G3FMJ5U","schema_version":"1.0","canonical_sha256":"e1b65627b4e8dc536e4efe20b150ff9b6d8210ddcf1df7b56473cbe733c8939b","source":{"kind":"arxiv","id":"2509.09174","version":1},"attestation_state":"computed","paper":{"title":"EchoX: Towards Mitigating Acoustic-Semantic Gap via Echo Training for Speech-to-Speech LLMs","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.SD"],"primary_cat":"cs.CL","authors_text":"Benyou Wang, Haizhou Li, Kaiqi Kou, Xiangnan Ma, Yuhao Du, Yuhao Zhang, Zhanchen Dai","submitted_at":"2025-09-11T06:17:59Z","abstract_excerpt":"Speech-to-speech large language models (SLLMs) are attracting increasing attention. Derived from text-based large language models (LLMs), SLLMs often exhibit degradation in knowledge and reasoning capabilities. We hypothesize that this limitation arises because current training paradigms for SLLMs fail to bridge the acoustic-semantic gap in the feature representation space. To address this issue, we propose EchoX, which leverages semantic representations and dynamically generates speech training targets. This approach integrates both acoustic and semantic learning, enabling EchoX to preserve s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.09174","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-09-11T06:17:59Z","cross_cats_sorted":["cs.AI","cs.SD"],"title_canon_sha256":"99b14337303b0e1b1a3d63eaabb7d6a9dd17095c72fb6e8329e2317c0a77f364","abstract_canon_sha256":"bac59efcea086f4d1444f80a8ea4471dd48fc609d563382a187191d09f8ab89c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:09:30.921096Z","signature_b64":"mEnLrmXHD6notnf33gbFiSIz4mZPO4qPfr9WGwDSc9XMkWZfx5WcEnX32Ce5iPmI60Qw3MFB8yohkVJl7dS4Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e1b65627b4e8dc536e4efe20b150ff9b6d8210ddcf1df7b56473cbe733c8939b","last_reissued_at":"2026-07-05T12:09:30.920597Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:09:30.920597Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EchoX: Towards Mitigating Acoustic-Semantic Gap via Echo Training for Speech-to-Speech LLMs","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.SD"],"primary_cat":"cs.CL","authors_text":"Benyou Wang, Haizhou Li, Kaiqi Kou, Xiangnan Ma, Yuhao Du, Yuhao Zhang, Zhanchen Dai","submitted_at":"2025-09-11T06:17:59Z","abstract_excerpt":"Speech-to-speech large language models (SLLMs) are attracting increasing attention. Derived from text-based large language models (LLMs), SLLMs often exhibit degradation in knowledge and reasoning capabilities. We hypothesize that this limitation arises because current training paradigms for SLLMs fail to bridge the acoustic-semantic gap in the feature representation space. To address this issue, we propose EchoX, which leverages semantic representations and dynamically generates speech training targets. This approach integrates both acoustic and semantic learning, enabling EchoX to preserve s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.09174","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.09174/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.09174","created_at":"2026-07-05T12:09:30.920662+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.09174v1","created_at":"2026-07-05T12:09:30.920662+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.09174","created_at":"2026-07-05T12:09:30.920662+00:00"},{"alias_kind":"pith_short_12","alias_value":"4G3FMJ5U5DOF","created_at":"2026-07-05T12:09:30.920662+00:00"},{"alias_kind":"pith_short_16","alias_value":"4G3FMJ5U5DOFG3SO","created_at":"2026-07-05T12:09:30.920662+00:00"},{"alias_kind":"pith_short_8","alias_value":"4G3FMJ5U","created_at":"2026-07-05T12:09:30.920662+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.08067","citing_title":"DialectS2S: End-to-End Speech Dialogue Modeling for Low-Resource Chinese Dialects","ref_index":24,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4G3FMJ5U5DOFG3SO7YQLCUH7TN","json":"https://pith.science/pith/4G3FMJ5U5DOFG3SO7YQLCUH7TN.json","graph_json":"https://pith.science/api/pith-number/4G3FMJ5U5DOFG3SO7YQLCUH7TN/graph.json","events_json":"https://pith.science/api/pith-number/4G3FMJ5U5DOFG3SO7YQLCUH7TN/events.json","paper":"https://pith.science/paper/4G3FMJ5U"},"agent_actions":{"view_html":"https://pith.science/pith/4G3FMJ5U5DOFG3SO7YQLCUH7TN","download_json":"https://pith.science/pith/4G3FMJ5U5DOFG3SO7YQLCUH7TN.json","view_paper":"https://pith.science/paper/4G3FMJ5U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.09174&json=true","fetch_graph":"https://pith.science/api/pith-number/4G3FMJ5U5DOFG3SO7YQLCUH7TN/graph.json","fetch_events":"https://pith.science/api/pith-number/4G3FMJ5U5DOFG3SO7YQLCUH7TN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4G3FMJ5U5DOFG3SO7YQLCUH7TN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4G3FMJ5U5DOFG3SO7YQLCUH7TN/action/storage_attestation","attest_author":"https://pith.science/pith/4G3FMJ5U5DOFG3SO7YQLCUH7TN/action/author_attestation","sign_citation":"https://pith.science/pith/4G3FMJ5U5DOFG3SO7YQLCUH7TN/action/citation_signature","submit_replication":"https://pith.science/pith/4G3FMJ5U5DOFG3SO7YQLCUH7TN/action/replication_record"}},"created_at":"2026-07-05T12:09:30.920662+00:00","updated_at":"2026-07-05T12:09:30.920662+00:00"}