{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MDUXQF64SVTZP742OM5P4RYRFC","short_pith_number":"pith:MDUXQF64","schema_version":"1.0","canonical_sha256":"60e97817dc956797ff9a733afe471128b1f0d504e3493bcfebf0f663189eb142","source":{"kind":"arxiv","id":"2304.06818","version":1},"attestation_state":"computed","paper":{"title":"Soundini: Sound-Guided Diffusion for Natural Video Editing","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Donghyeon Cho, Feng Yang, Huiwen Chang, Innfarn Yoo, Jinkyu Kim, Sangpil Kim, Seung Hyun Lee, Sieun Kim, Youngseo Kim","submitted_at":"2023-04-13T20:56:53Z","abstract_excerpt":"We propose a method for adding sound-guided visual effects to specific regions of videos with a zero-shot setting. Animating the appearance of the visual effect is challenging because each frame of the edited video should have visual changes while maintaining temporal consistency. Moreover, existing video editing solutions focus on temporal consistency across frames, ignoring the visual style variations over time, e.g., thunderstorm, wave, fire crackling. To overcome this limitation, we utilize temporal sound features for the dynamic style. Specifically, we guide denoising diffusion probabilis"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.06818","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2023-04-13T20:56:53Z","cross_cats_sorted":[],"title_canon_sha256":"4b3ba9f093da87f2aea873efb180f29ac082cdbb41cf516c8e99a23b4ef34915","abstract_canon_sha256":"b322bc847deb64322ce6dc61062edaef887c143736e7498b46ddfb751d904c66"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:01:02.363347Z","signature_b64":"Ad5was71GT8Vyeolo4TmQM2VrN71SmIk7CdSI2N1XpV50atCjwOhL6Zm8S39tb6FHZR2Th28mxfHVRr23qZRCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"60e97817dc956797ff9a733afe471128b1f0d504e3493bcfebf0f663189eb142","last_reissued_at":"2026-07-05T06:01:02.362863Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:01:02.362863Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Soundini: Sound-Guided Diffusion for Natural Video Editing","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Donghyeon Cho, Feng Yang, Huiwen Chang, Innfarn Yoo, Jinkyu Kim, Sangpil Kim, Seung Hyun Lee, Sieun Kim, Youngseo Kim","submitted_at":"2023-04-13T20:56:53Z","abstract_excerpt":"We propose a method for adding sound-guided visual effects to specific regions of videos with a zero-shot setting. Animating the appearance of the visual effect is challenging because each frame of the edited video should have visual changes while maintaining temporal consistency. Moreover, existing video editing solutions focus on temporal consistency across frames, ignoring the visual style variations over time, e.g., thunderstorm, wave, fire crackling. To overcome this limitation, we utilize temporal sound features for the dynamic style. Specifically, we guide denoising diffusion probabilis"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.06818","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.06818/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.06818","created_at":"2026-07-05T06:01:02.362921+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.06818v1","created_at":"2026-07-05T06:01:02.362921+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.06818","created_at":"2026-07-05T06:01:02.362921+00:00"},{"alias_kind":"pith_short_12","alias_value":"MDUXQF64SVTZ","created_at":"2026-07-05T06:01:02.362921+00:00"},{"alias_kind":"pith_short_16","alias_value":"MDUXQF64SVTZP742","created_at":"2026-07-05T06:01:02.362921+00:00"},{"alias_kind":"pith_short_8","alias_value":"MDUXQF64","created_at":"2026-07-05T06:01:02.362921+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.10571","citing_title":"AVI-Edit: Audio-sync Video Instance Editing with Granularity-Aware Mask Refiner","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MDUXQF64SVTZP742OM5P4RYRFC","json":"https://pith.science/pith/MDUXQF64SVTZP742OM5P4RYRFC.json","graph_json":"https://pith.science/api/pith-number/MDUXQF64SVTZP742OM5P4RYRFC/graph.json","events_json":"https://pith.science/api/pith-number/MDUXQF64SVTZP742OM5P4RYRFC/events.json","paper":"https://pith.science/paper/MDUXQF64"},"agent_actions":{"view_html":"https://pith.science/pith/MDUXQF64SVTZP742OM5P4RYRFC","download_json":"https://pith.science/pith/MDUXQF64SVTZP742OM5P4RYRFC.json","view_paper":"https://pith.science/paper/MDUXQF64","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.06818&json=true","fetch_graph":"https://pith.science/api/pith-number/MDUXQF64SVTZP742OM5P4RYRFC/graph.json","fetch_events":"https://pith.science/api/pith-number/MDUXQF64SVTZP742OM5P4RYRFC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MDUXQF64SVTZP742OM5P4RYRFC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MDUXQF64SVTZP742OM5P4RYRFC/action/storage_attestation","attest_author":"https://pith.science/pith/MDUXQF64SVTZP742OM5P4RYRFC/action/author_attestation","sign_citation":"https://pith.science/pith/MDUXQF64SVTZP742OM5P4RYRFC/action/citation_signature","submit_replication":"https://pith.science/pith/MDUXQF64SVTZP742OM5P4RYRFC/action/replication_record"}},"created_at":"2026-07-05T06:01:02.362921+00:00","updated_at":"2026-07-05T06:01:02.362921+00:00"}