{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YO3IO7F2DDX4SRYS2QHA6D5T7M","short_pith_number":"pith:YO3IO7F2","schema_version":"1.0","canonical_sha256":"c3b6877cba18efc94712d40e0f0fb3fb263e552c75a043560d8f7d4804575b0a","source":{"kind":"arxiv","id":"2508.07302","version":2},"attestation_state":"computed","paper":{"title":"XEmoRAG: Cross-Lingual Emotion Transfer with Controllable Intensity Using Retrieval-Augmented Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Danming Xie, Hai Li, Jingbin Hu, Junhui Liu, Lei Xie, Tianlun Zuo, Xinfa Zhu, Ying Yan, Yuke Li","submitted_at":"2025-08-10T11:31:13Z","abstract_excerpt":"Zero-shot emotion transfer in cross-lingual speech synthesis refers to generating speech in a target language, where the emotion is expressed based on reference speech from a different source language. However, this task remains challenging due to the scarcity of parallel multilingual emotional corpora, the presence of foreign accent artifacts, and the difficulty of separating emotion from language-specific prosodic features. In this paper, we propose XEmoRAG, a novel framework to enable zero-shot emotion transfer from Chinese to Thai using a large language model (LLM)-based model, without rel"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.07302","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2025-08-10T11:31:13Z","cross_cats_sorted":[],"title_canon_sha256":"0b934308be8941b11ae79cd0f497b04afd6abbd7698d2dba3aaace8cd8833598","abstract_canon_sha256":"1ff6d3604d3e8c35a55cb4a4b27517a37974c7b064cadc1a6f45a236655fc357"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:52:24.275969Z","signature_b64":"AyfrsISzgNvdDQlV/DFbUS+zo33lx+9Hwy66Nv4FZGvBON+e+4dWW06xqt3ucdAdG7G3mduhHPX84arQm+wVAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c3b6877cba18efc94712d40e0f0fb3fb263e552c75a043560d8f7d4804575b0a","last_reissued_at":"2026-07-05T11:52:24.275455Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:52:24.275455Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"XEmoRAG: Cross-Lingual Emotion Transfer with Controllable Intensity Using Retrieval-Augmented Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Danming Xie, Hai Li, Jingbin Hu, Junhui Liu, Lei Xie, Tianlun Zuo, Xinfa Zhu, Ying Yan, Yuke Li","submitted_at":"2025-08-10T11:31:13Z","abstract_excerpt":"Zero-shot emotion transfer in cross-lingual speech synthesis refers to generating speech in a target language, where the emotion is expressed based on reference speech from a different source language. However, this task remains challenging due to the scarcity of parallel multilingual emotional corpora, the presence of foreign accent artifacts, and the difficulty of separating emotion from language-specific prosodic features. In this paper, we propose XEmoRAG, a novel framework to enable zero-shot emotion transfer from Chinese to Thai using a large language model (LLM)-based model, without rel"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.07302","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.07302/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.07302","created_at":"2026-07-05T11:52:24.275520+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.07302v2","created_at":"2026-07-05T11:52:24.275520+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.07302","created_at":"2026-07-05T11:52:24.275520+00:00"},{"alias_kind":"pith_short_12","alias_value":"YO3IO7F2DDX4","created_at":"2026-07-05T11:52:24.275520+00:00"},{"alias_kind":"pith_short_16","alias_value":"YO3IO7F2DDX4SRYS","created_at":"2026-07-05T11:52:24.275520+00:00"},{"alias_kind":"pith_short_8","alias_value":"YO3IO7F2","created_at":"2026-07-05T11:52:24.275520+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.14804","citing_title":"Towards Building Speech Large Language Models for Multitask Understanding in Low-Resource Languages","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YO3IO7F2DDX4SRYS2QHA6D5T7M","json":"https://pith.science/pith/YO3IO7F2DDX4SRYS2QHA6D5T7M.json","graph_json":"https://pith.science/api/pith-number/YO3IO7F2DDX4SRYS2QHA6D5T7M/graph.json","events_json":"https://pith.science/api/pith-number/YO3IO7F2DDX4SRYS2QHA6D5T7M/events.json","paper":"https://pith.science/paper/YO3IO7F2"},"agent_actions":{"view_html":"https://pith.science/pith/YO3IO7F2DDX4SRYS2QHA6D5T7M","download_json":"https://pith.science/pith/YO3IO7F2DDX4SRYS2QHA6D5T7M.json","view_paper":"https://pith.science/paper/YO3IO7F2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.07302&json=true","fetch_graph":"https://pith.science/api/pith-number/YO3IO7F2DDX4SRYS2QHA6D5T7M/graph.json","fetch_events":"https://pith.science/api/pith-number/YO3IO7F2DDX4SRYS2QHA6D5T7M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YO3IO7F2DDX4SRYS2QHA6D5T7M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YO3IO7F2DDX4SRYS2QHA6D5T7M/action/storage_attestation","attest_author":"https://pith.science/pith/YO3IO7F2DDX4SRYS2QHA6D5T7M/action/author_attestation","sign_citation":"https://pith.science/pith/YO3IO7F2DDX4SRYS2QHA6D5T7M/action/citation_signature","submit_replication":"https://pith.science/pith/YO3IO7F2DDX4SRYS2QHA6D5T7M/action/replication_record"}},"created_at":"2026-07-05T11:52:24.275520+00:00","updated_at":"2026-07-05T11:52:24.275520+00:00"}