{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:FZ2NKRGG2JK5LTLVOFCSG3URRE","short_pith_number":"pith:FZ2NKRGG","schema_version":"1.0","canonical_sha256":"2e74d544c6d255d5cd757145236e918918482fb0b2629b74962983d623998d30","source":{"kind":"arxiv","id":"1908.07094","version":1},"attestation_state":"computed","paper":{"title":"Unpaired Image-to-Speech Synthesis with Multimodal Information Bottleneck","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Daniel McDuff, Shuang Ma, Yale Song","submitted_at":"2019-08-19T22:52:59Z","abstract_excerpt":"Deep generative models have led to significant advances in cross-modal generation such as text-to-image synthesis. Training these models typically requires paired data with direct correspondence between modalities. We introduce the novel problem of translating instances from one modality to another without paired data by leveraging an intermediate modality shared by the two other modalities. To demonstrate this, we take the problem of translating images to speech. In this case, one could leverage disjoint datasets with one shared modality, e.g., image-text pairs and text-speech pairs, with tex"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1908.07094","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2019-08-19T22:52:59Z","cross_cats_sorted":[],"title_canon_sha256":"d15c24586c05148edea41ad16a3628ca57269636206c0101a2056aedce6c2d33","abstract_canon_sha256":"7e3addb9db4ec1d89d5560335d4d61101e57fa3352d0de0c70899000939115b3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-04T23:58:30.715434Z","signature_b64":"XA/LFt5okuNydXh9ur41HD/oaJOffLnTMAdxPwdnsoFqdueuBscWtsF3i+VqAI90m3Xpl9jwz8yAUmLSFwG5DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2e74d544c6d255d5cd757145236e918918482fb0b2629b74962983d623998d30","last_reissued_at":"2026-07-04T23:58:30.715034Z","signature_status":"signed_v1","first_computed_at":"2026-07-04T23:58:30.715034Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unpaired Image-to-Speech Synthesis with Multimodal Information Bottleneck","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Daniel McDuff, Shuang Ma, Yale Song","submitted_at":"2019-08-19T22:52:59Z","abstract_excerpt":"Deep generative models have led to significant advances in cross-modal generation such as text-to-image synthesis. Training these models typically requires paired data with direct correspondence between modalities. We introduce the novel problem of translating instances from one modality to another without paired data by leveraging an intermediate modality shared by the two other modalities. To demonstrate this, we take the problem of translating images to speech. In this case, one could leverage disjoint datasets with one shared modality, e.g., image-text pairs and text-speech pairs, with tex"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.07094","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1908.07094/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1908.07094","created_at":"2026-07-04T23:58:30.715091+00:00"},{"alias_kind":"arxiv_version","alias_value":"1908.07094v1","created_at":"2026-07-04T23:58:30.715091+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.07094","created_at":"2026-07-04T23:58:30.715091+00:00"},{"alias_kind":"pith_short_12","alias_value":"FZ2NKRGG2JK5","created_at":"2026-07-04T23:58:30.715091+00:00"},{"alias_kind":"pith_short_16","alias_value":"FZ2NKRGG2JK5LTLV","created_at":"2026-07-04T23:58:30.715091+00:00"},{"alias_kind":"pith_short_8","alias_value":"FZ2NKRGG","created_at":"2026-07-04T23:58:30.715091+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FZ2NKRGG2JK5LTLVOFCSG3URRE","json":"https://pith.science/pith/FZ2NKRGG2JK5LTLVOFCSG3URRE.json","graph_json":"https://pith.science/api/pith-number/FZ2NKRGG2JK5LTLVOFCSG3URRE/graph.json","events_json":"https://pith.science/api/pith-number/FZ2NKRGG2JK5LTLVOFCSG3URRE/events.json","paper":"https://pith.science/paper/FZ2NKRGG"},"agent_actions":{"view_html":"https://pith.science/pith/FZ2NKRGG2JK5LTLVOFCSG3URRE","download_json":"https://pith.science/pith/FZ2NKRGG2JK5LTLVOFCSG3URRE.json","view_paper":"https://pith.science/paper/FZ2NKRGG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1908.07094&json=true","fetch_graph":"https://pith.science/api/pith-number/FZ2NKRGG2JK5LTLVOFCSG3URRE/graph.json","fetch_events":"https://pith.science/api/pith-number/FZ2NKRGG2JK5LTLVOFCSG3URRE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FZ2NKRGG2JK5LTLVOFCSG3URRE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FZ2NKRGG2JK5LTLVOFCSG3URRE/action/storage_attestation","attest_author":"https://pith.science/pith/FZ2NKRGG2JK5LTLVOFCSG3URRE/action/author_attestation","sign_citation":"https://pith.science/pith/FZ2NKRGG2JK5LTLVOFCSG3URRE/action/citation_signature","submit_replication":"https://pith.science/pith/FZ2NKRGG2JK5LTLVOFCSG3URRE/action/replication_record"}},"created_at":"2026-07-04T23:58:30.715091+00:00","updated_at":"2026-07-04T23:58:30.715091+00:00"}