{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:XN6A7MS3E3VOUHSB4KKCFONU7L","short_pith_number":"pith:XN6A7MS3","schema_version":"1.0","canonical_sha256":"bb7c0fb25b26eaea1e41e29422b9b4fad01fc6bc32c4da3a8dc6e6c86e43190e","source":{"kind":"arxiv","id":"2104.06900","version":1},"attestation_state":"computed","paper":{"title":"FastS2S-VC: Streaming Non-Autoregressive Sequence-to-Sequence Voice Conversion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Hirokazu Kameoka, Kou Tanaka, Takuhiro Kaneko","submitted_at":"2021-04-14T14:48:22Z","abstract_excerpt":"This paper proposes a non-autoregressive extension of our previously proposed sequence-to-sequence (S2S) model-based voice conversion (VC) methods. S2S model-based VC methods have attracted particular attention in recent years for their flexibility in converting not only the voice identity but also the pitch contour and local duration of input speech, thanks to the ability of the encoder-decoder architecture with the attention mechanism. However, one of the obstacles to making these methods work in real-time is the autoregressive (AR) structure. To overcome this obstacle, we develop a method t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.06900","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2021-04-14T14:48:22Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"8234b0518e509f7b804795fdc72af44352668741e21eca2525325289ffb48108","abstract_canon_sha256":"a5f4c9dc1641ac2e34c6cd927112a42de0930d60c3f8fccaff1e4a5b4ede2986"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:32:10.829379Z","signature_b64":"/L1VE/w1YVSK9/dqt7RwsB5kcZka1vk4z8C0UqdbuGsDKMe+z7yOJkc4LBpZAQOGqhtj/mZ7OauyGiuKwQWeBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb7c0fb25b26eaea1e41e29422b9b4fad01fc6bc32c4da3a8dc6e6c86e43190e","last_reissued_at":"2026-07-05T02:32:10.828849Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:32:10.828849Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FastS2S-VC: Streaming Non-Autoregressive Sequence-to-Sequence Voice Conversion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Hirokazu Kameoka, Kou Tanaka, Takuhiro Kaneko","submitted_at":"2021-04-14T14:48:22Z","abstract_excerpt":"This paper proposes a non-autoregressive extension of our previously proposed sequence-to-sequence (S2S) model-based voice conversion (VC) methods. S2S model-based VC methods have attracted particular attention in recent years for their flexibility in converting not only the voice identity but also the pitch contour and local duration of input speech, thanks to the ability of the encoder-decoder architecture with the attention mechanism. However, one of the obstacles to making these methods work in real-time is the autoregressive (AR) structure. To overcome this obstacle, we develop a method t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.06900","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.06900/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.06900","created_at":"2026-07-05T02:32:10.828920+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.06900v1","created_at":"2026-07-05T02:32:10.828920+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.06900","created_at":"2026-07-05T02:32:10.828920+00:00"},{"alias_kind":"pith_short_12","alias_value":"XN6A7MS3E3VO","created_at":"2026-07-05T02:32:10.828920+00:00"},{"alias_kind":"pith_short_16","alias_value":"XN6A7MS3E3VOUHSB","created_at":"2026-07-05T02:32:10.828920+00:00"},{"alias_kind":"pith_short_8","alias_value":"XN6A7MS3","created_at":"2026-07-05T02:32:10.828920+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01905","citing_title":"Advancing Electrolaryngeal Speech Enhancement Through Speech-Text Representation Learning","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XN6A7MS3E3VOUHSB4KKCFONU7L","json":"https://pith.science/pith/XN6A7MS3E3VOUHSB4KKCFONU7L.json","graph_json":"https://pith.science/api/pith-number/XN6A7MS3E3VOUHSB4KKCFONU7L/graph.json","events_json":"https://pith.science/api/pith-number/XN6A7MS3E3VOUHSB4KKCFONU7L/events.json","paper":"https://pith.science/paper/XN6A7MS3"},"agent_actions":{"view_html":"https://pith.science/pith/XN6A7MS3E3VOUHSB4KKCFONU7L","download_json":"https://pith.science/pith/XN6A7MS3E3VOUHSB4KKCFONU7L.json","view_paper":"https://pith.science/paper/XN6A7MS3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.06900&json=true","fetch_graph":"https://pith.science/api/pith-number/XN6A7MS3E3VOUHSB4KKCFONU7L/graph.json","fetch_events":"https://pith.science/api/pith-number/XN6A7MS3E3VOUHSB4KKCFONU7L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XN6A7MS3E3VOUHSB4KKCFONU7L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XN6A7MS3E3VOUHSB4KKCFONU7L/action/storage_attestation","attest_author":"https://pith.science/pith/XN6A7MS3E3VOUHSB4KKCFONU7L/action/author_attestation","sign_citation":"https://pith.science/pith/XN6A7MS3E3VOUHSB4KKCFONU7L/action/citation_signature","submit_replication":"https://pith.science/pith/XN6A7MS3E3VOUHSB4KKCFONU7L/action/replication_record"}},"created_at":"2026-07-05T02:32:10.828920+00:00","updated_at":"2026-07-05T02:32:10.828920+00:00"}