{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TUGCXCLSSX6CZYQZKUKELMARJD","short_pith_number":"pith:TUGCXCLS","schema_version":"1.0","canonical_sha256":"9d0c2b897295fc2ce219551445b01148c3429b54c57d45ba5496e107185746dc","source":{"kind":"arxiv","id":"2501.06394","version":1},"attestation_state":"computed","paper":{"title":"Unispeaker: A Unified Approach for Multimodality-driven Speaker Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Heng Lu, Shiliang Zhang, Zhengyan Sheng, Zhen-Hua Ling, Zhihao Du","submitted_at":"2025-01-11T00:47:29Z","abstract_excerpt":"Recent advancements in personalized speech generation have brought synthetic speech increasingly close to the realism of target speakers' recordings, yet multimodal speaker generation remains on the rise. This paper introduces UniSpeaker, a unified approach for multimodality-driven speaker generation. Specifically, we propose a unified voice aggregator based on KV-Former, applying soft contrastive loss to map diverse voice description modalities into a shared voice space, ensuring that the generated voice aligns more closely with the input descriptions. To evaluate multimodality-driven voice c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.06394","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2025-01-11T00:47:29Z","cross_cats_sorted":["cs.AI","eess.AS"],"title_canon_sha256":"3afb4ddb1e5f50b4a47de6639c40ce7cae3e31b9bb3d42c37d71503143323ad6","abstract_canon_sha256":"14cad791323becfb5b28e3e6da584b82463e715f8a2ac837f97982d506b42053"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:02.626392Z","signature_b64":"u3il11S1wOJEIgglVfSOBdEsIcfIdev5BICTDaRh3zSqMycBdDmzcnDiAYA8pQFIMGFTRjnEYIBZnTflp4eCBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d0c2b897295fc2ce219551445b01148c3429b54c57d45ba5496e107185746dc","last_reissued_at":"2026-07-05T10:00:02.625865Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:02.625865Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unispeaker: A Unified Approach for Multimodality-driven Speaker Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Heng Lu, Shiliang Zhang, Zhengyan Sheng, Zhen-Hua Ling, Zhihao Du","submitted_at":"2025-01-11T00:47:29Z","abstract_excerpt":"Recent advancements in personalized speech generation have brought synthetic speech increasingly close to the realism of target speakers' recordings, yet multimodal speaker generation remains on the rise. This paper introduces UniSpeaker, a unified approach for multimodality-driven speaker generation. Specifically, we propose a unified voice aggregator based on KV-Former, applying soft contrastive loss to map diverse voice description modalities into a shared voice space, ensuring that the generated voice aligns more closely with the input descriptions. To evaluate multimodality-driven voice c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.06394","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.06394/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.06394","created_at":"2026-07-05T10:00:02.625923+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.06394v1","created_at":"2026-07-05T10:00:02.625923+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.06394","created_at":"2026-07-05T10:00:02.625923+00:00"},{"alias_kind":"pith_short_12","alias_value":"TUGCXCLSSX6C","created_at":"2026-07-05T10:00:02.625923+00:00"},{"alias_kind":"pith_short_16","alias_value":"TUGCXCLSSX6CZYQZ","created_at":"2026-07-05T10:00:02.625923+00:00"},{"alias_kind":"pith_short_8","alias_value":"TUGCXCLS","created_at":"2026-07-05T10:00:02.625923+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.17589","citing_title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TUGCXCLSSX6CZYQZKUKELMARJD","json":"https://pith.science/pith/TUGCXCLSSX6CZYQZKUKELMARJD.json","graph_json":"https://pith.science/api/pith-number/TUGCXCLSSX6CZYQZKUKELMARJD/graph.json","events_json":"https://pith.science/api/pith-number/TUGCXCLSSX6CZYQZKUKELMARJD/events.json","paper":"https://pith.science/paper/TUGCXCLS"},"agent_actions":{"view_html":"https://pith.science/pith/TUGCXCLSSX6CZYQZKUKELMARJD","download_json":"https://pith.science/pith/TUGCXCLSSX6CZYQZKUKELMARJD.json","view_paper":"https://pith.science/paper/TUGCXCLS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.06394&json=true","fetch_graph":"https://pith.science/api/pith-number/TUGCXCLSSX6CZYQZKUKELMARJD/graph.json","fetch_events":"https://pith.science/api/pith-number/TUGCXCLSSX6CZYQZKUKELMARJD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TUGCXCLSSX6CZYQZKUKELMARJD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TUGCXCLSSX6CZYQZKUKELMARJD/action/storage_attestation","attest_author":"https://pith.science/pith/TUGCXCLSSX6CZYQZKUKELMARJD/action/author_attestation","sign_citation":"https://pith.science/pith/TUGCXCLSSX6CZYQZKUKELMARJD/action/citation_signature","submit_replication":"https://pith.science/pith/TUGCXCLSSX6CZYQZKUKELMARJD/action/replication_record"}},"created_at":"2026-07-05T10:00:02.625923+00:00","updated_at":"2026-07-05T10:00:02.625923+00:00"}