{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2VPJWADTMH4CQ4DAEXRODPQZ2U","short_pith_number":"pith:2VPJWADT","schema_version":"1.0","canonical_sha256":"d55e9b007361f828706025e2e1be19d5144d4acdb146de11a6ffb038969ff92f","source":{"kind":"arxiv","id":"2406.12688","version":1},"attestation_state":"computed","paper":{"title":"Speak in the Scene: Diffusion-based Acoustic Scene Transfer toward Immersive Speech Generation","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["eess.SP"],"primary_cat":"eess.AS","authors_text":"Hong-Goo Kang, Min-Seok Choi, Miseul Kim, Soo-Whan Chung, Youna Ji","submitted_at":"2024-06-18T15:00:25Z","abstract_excerpt":"This paper introduces a novel task in generative speech processing, Acoustic Scene Transfer (AST), which aims to transfer acoustic scenes of speech signals to diverse environments. AST promises an immersive experience in speech perception by adapting the acoustic scene behind speech signals to desired environments. We propose AST-LDM for the AST task, which generates speech signals accompanied by the target acoustic scene of the reference prompt. Specifically, AST-LDM is a latent diffusion model conditioned by CLAP embeddings that describe target acoustic scenes in either audio or text modalit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.12688","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"eess.AS","submitted_at":"2024-06-18T15:00:25Z","cross_cats_sorted":["eess.SP"],"title_canon_sha256":"3a307b60c548b7b85f5239c756bc553ce87a0fc4578505aff778eea860d53ca1","abstract_canon_sha256":"dd5998369365e283b94a6bfb9ab96fb6938b50eb6132c98f9fea0ef3a1943e71"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:33:50.014772Z","signature_b64":"4K4XF8gsG4uh0FA0vFfTBdtJAAZW3DvzWNBV8JxOIp7/lNHQCGzLrMdzF678tUYmYLWV0B0ZlAL2yNztVNldAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d55e9b007361f828706025e2e1be19d5144d4acdb146de11a6ffb038969ff92f","last_reissued_at":"2026-07-05T08:33:50.014374Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:33:50.014374Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Speak in the Scene: Diffusion-based Acoustic Scene Transfer toward Immersive Speech Generation","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["eess.SP"],"primary_cat":"eess.AS","authors_text":"Hong-Goo Kang, Min-Seok Choi, Miseul Kim, Soo-Whan Chung, Youna Ji","submitted_at":"2024-06-18T15:00:25Z","abstract_excerpt":"This paper introduces a novel task in generative speech processing, Acoustic Scene Transfer (AST), which aims to transfer acoustic scenes of speech signals to diverse environments. AST promises an immersive experience in speech perception by adapting the acoustic scene behind speech signals to desired environments. We propose AST-LDM for the AST task, which generates speech signals accompanied by the target acoustic scene of the reference prompt. Specifically, AST-LDM is a latent diffusion model conditioned by CLAP embeddings that describe target acoustic scenes in either audio or text modalit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.12688","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.12688/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.12688","created_at":"2026-07-05T08:33:50.014430+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.12688v1","created_at":"2026-07-05T08:33:50.014430+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12688","created_at":"2026-07-05T08:33:50.014430+00:00"},{"alias_kind":"pith_short_12","alias_value":"2VPJWADTMH4C","created_at":"2026-07-05T08:33:50.014430+00:00"},{"alias_kind":"pith_short_16","alias_value":"2VPJWADTMH4CQ4DA","created_at":"2026-07-05T08:33:50.014430+00:00"},{"alias_kind":"pith_short_8","alias_value":"2VPJWADT","created_at":"2026-07-05T08:33:50.014430+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.11283","citing_title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","ref_index":99,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2VPJWADTMH4CQ4DAEXRODPQZ2U","json":"https://pith.science/pith/2VPJWADTMH4CQ4DAEXRODPQZ2U.json","graph_json":"https://pith.science/api/pith-number/2VPJWADTMH4CQ4DAEXRODPQZ2U/graph.json","events_json":"https://pith.science/api/pith-number/2VPJWADTMH4CQ4DAEXRODPQZ2U/events.json","paper":"https://pith.science/paper/2VPJWADT"},"agent_actions":{"view_html":"https://pith.science/pith/2VPJWADTMH4CQ4DAEXRODPQZ2U","download_json":"https://pith.science/pith/2VPJWADTMH4CQ4DAEXRODPQZ2U.json","view_paper":"https://pith.science/paper/2VPJWADT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.12688&json=true","fetch_graph":"https://pith.science/api/pith-number/2VPJWADTMH4CQ4DAEXRODPQZ2U/graph.json","fetch_events":"https://pith.science/api/pith-number/2VPJWADTMH4CQ4DAEXRODPQZ2U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2VPJWADTMH4CQ4DAEXRODPQZ2U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2VPJWADTMH4CQ4DAEXRODPQZ2U/action/storage_attestation","attest_author":"https://pith.science/pith/2VPJWADTMH4CQ4DAEXRODPQZ2U/action/author_attestation","sign_citation":"https://pith.science/pith/2VPJWADTMH4CQ4DAEXRODPQZ2U/action/citation_signature","submit_replication":"https://pith.science/pith/2VPJWADTMH4CQ4DAEXRODPQZ2U/action/replication_record"}},"created_at":"2026-07-05T08:33:50.014430+00:00","updated_at":"2026-07-05T08:33:50.014430+00:00"}