{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:GDKM73RVQVIZ5N7LSEH5NJJZ3L","short_pith_number":"pith:GDKM73RV","schema_version":"1.0","canonical_sha256":"30d4cfee3585519eb7eb910fd6a539dac65f0884d65f6c19426d8eeed976dc22","source":{"kind":"arxiv","id":"2010.12973","version":2},"attestation_state":"computed","paper":{"title":"Unsupervised Learning of Disentangled Speech Content and Style Representation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Andros Tjandra, Ruoming Pang, Shigeki Karita, Yu Zhang","submitted_at":"2020-10-24T20:16:03Z","abstract_excerpt":"We present an approach for unsupervised learning of speech representation disentangling contents and styles. Our model consists of: (1) a local encoder that captures per-frame information; (2) a global encoder that captures per-utterance information; and (3) a conditional decoder that reconstructs speech given local and global latent variables. Our experiments show that (1) the local latent variables encode speech contents, as reconstructed speech can be recognized by ASR with low word error rates (WER), even with a different global encoding; (2) the global latent variables encode speaker styl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.12973","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-10-24T20:16:03Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"5e495399255f57507fb96655b5fbb969071205296197dd7c1234508141e88d6f","abstract_canon_sha256":"42a072f2a1fafd38061ebbca5a46d2378cd993f52b7aaf14444d55b9cb5fdc95"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:50:41.481436Z","signature_b64":"4XDG8xySWvqHPGg//06KLie0BQRKcMKRmPtPJQtyj2jX56Ciw4UxI9l/B663Mo4vz9QrcCPGZjw0y4dA0MIGCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"30d4cfee3585519eb7eb910fd6a539dac65f0884d65f6c19426d8eeed976dc22","last_reissued_at":"2026-07-05T02:50:41.480938Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:50:41.480938Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unsupervised Learning of Disentangled Speech Content and Style Representation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Andros Tjandra, Ruoming Pang, Shigeki Karita, Yu Zhang","submitted_at":"2020-10-24T20:16:03Z","abstract_excerpt":"We present an approach for unsupervised learning of speech representation disentangling contents and styles. Our model consists of: (1) a local encoder that captures per-frame information; (2) a global encoder that captures per-utterance information; and (3) a conditional decoder that reconstructs speech given local and global latent variables. Our experiments show that (1) the local latent variables encode speech contents, as reconstructed speech can be recognized by ASR with low word error rates (WER), even with a different global encoding; (2) the global latent variables encode speaker styl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.12973","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.12973/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.12973","created_at":"2026-07-05T02:50:41.481001+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.12973v2","created_at":"2026-07-05T02:50:41.481001+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.12973","created_at":"2026-07-05T02:50:41.481001+00:00"},{"alias_kind":"pith_short_12","alias_value":"GDKM73RVQVIZ","created_at":"2026-07-05T02:50:41.481001+00:00"},{"alias_kind":"pith_short_16","alias_value":"GDKM73RVQVIZ5N7L","created_at":"2026-07-05T02:50:41.481001+00:00"},{"alias_kind":"pith_short_8","alias_value":"GDKM73RV","created_at":"2026-07-05T02:50:41.481001+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.04817","citing_title":"Fast-VGAN: Lightweight Voice Conversion with Explicit Control of F0 and Duration Parameters","ref_index":26,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GDKM73RVQVIZ5N7LSEH5NJJZ3L","json":"https://pith.science/pith/GDKM73RVQVIZ5N7LSEH5NJJZ3L.json","graph_json":"https://pith.science/api/pith-number/GDKM73RVQVIZ5N7LSEH5NJJZ3L/graph.json","events_json":"https://pith.science/api/pith-number/GDKM73RVQVIZ5N7LSEH5NJJZ3L/events.json","paper":"https://pith.science/paper/GDKM73RV"},"agent_actions":{"view_html":"https://pith.science/pith/GDKM73RVQVIZ5N7LSEH5NJJZ3L","download_json":"https://pith.science/pith/GDKM73RVQVIZ5N7LSEH5NJJZ3L.json","view_paper":"https://pith.science/paper/GDKM73RV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.12973&json=true","fetch_graph":"https://pith.science/api/pith-number/GDKM73RVQVIZ5N7LSEH5NJJZ3L/graph.json","fetch_events":"https://pith.science/api/pith-number/GDKM73RVQVIZ5N7LSEH5NJJZ3L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GDKM73RVQVIZ5N7LSEH5NJJZ3L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GDKM73RVQVIZ5N7LSEH5NJJZ3L/action/storage_attestation","attest_author":"https://pith.science/pith/GDKM73RVQVIZ5N7LSEH5NJJZ3L/action/author_attestation","sign_citation":"https://pith.science/pith/GDKM73RVQVIZ5N7LSEH5NJJZ3L/action/citation_signature","submit_replication":"https://pith.science/pith/GDKM73RVQVIZ5N7LSEH5NJJZ3L/action/replication_record"}},"created_at":"2026-07-05T02:50:41.481001+00:00","updated_at":"2026-07-05T02:50:41.481001+00:00"}