{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PTAQZQ7NJTJW34ME7R3UJIGUJW","short_pith_number":"pith:PTAQZQ7N","schema_version":"1.0","canonical_sha256":"7cc10cc3ed4cd36df184fc7744a0d44dbc1f08e2ec889ebc60922bac69e7acf5","source":{"kind":"arxiv","id":"2501.00645","version":1},"attestation_state":"computed","paper":{"title":"SoundBrush: Sound as a Brush for Visual Scene Editing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Junseok Ko, Kim Jun-Seong, Kim Sung-Bin, Tae-Hyun Oh, Yewon Kim","submitted_at":"2024-12-31T20:53:45Z","abstract_excerpt":"We propose SoundBrush, a model that uses sound as a brush to edit and manipulate visual scenes. We extend the generative capabilities of the Latent Diffusion Model (LDM) to incorporate audio information for editing visual scenes. Inspired by existing image-editing works, we frame this task as a supervised learning problem and leverage various off-the-shelf models to construct a sound-paired visual scene dataset for training. This richly generated dataset enables SoundBrush to learn to map audio features into the textual space of the LDM, allowing for visual scene editing guided by diverse in-t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.00645","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-31T20:53:45Z","cross_cats_sorted":["cs.LG","cs.SD","eess.AS"],"title_canon_sha256":"c4fd4588dc7b3c98db48587e0691ffa8f9c102acfe0eb083ca33f974c2f43621","abstract_canon_sha256":"514221cc31a026a78f1fc3c91ab8043682256e66633f340c64102448279ac0a6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:56:08.616521Z","signature_b64":"9MisQrWQBWxUF8qHHfEmvx4RZ1Xot8+gdOZ8dw7f/VMdMn1H5Ze8JzJDn64Fu6JUnGudEXQXI/tojqJjbglhDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7cc10cc3ed4cd36df184fc7744a0d44dbc1f08e2ec889ebc60922bac69e7acf5","last_reissued_at":"2026-07-05T09:56:08.616134Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:56:08.616134Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SoundBrush: Sound as a Brush for Visual Scene Editing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Junseok Ko, Kim Jun-Seong, Kim Sung-Bin, Tae-Hyun Oh, Yewon Kim","submitted_at":"2024-12-31T20:53:45Z","abstract_excerpt":"We propose SoundBrush, a model that uses sound as a brush to edit and manipulate visual scenes. We extend the generative capabilities of the Latent Diffusion Model (LDM) to incorporate audio information for editing visual scenes. Inspired by existing image-editing works, we frame this task as a supervised learning problem and leverage various off-the-shelf models to construct a sound-paired visual scene dataset for training. This richly generated dataset enables SoundBrush to learn to map audio features into the textual space of the LDM, allowing for visual scene editing guided by diverse in-t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.00645","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.00645/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.00645","created_at":"2026-07-05T09:56:08.616195+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.00645v1","created_at":"2026-07-05T09:56:08.616195+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.00645","created_at":"2026-07-05T09:56:08.616195+00:00"},{"alias_kind":"pith_short_12","alias_value":"PTAQZQ7NJTJW","created_at":"2026-07-05T09:56:08.616195+00:00"},{"alias_kind":"pith_short_16","alias_value":"PTAQZQ7NJTJW34ME","created_at":"2026-07-05T09:56:08.616195+00:00"},{"alias_kind":"pith_short_8","alias_value":"PTAQZQ7N","created_at":"2026-07-05T09:56:08.616195+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.18750","citing_title":"CatchPhrase: EXPrompt-Guided Encoder Adaptation for Audio-to-Image Generation","ref_index":45,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PTAQZQ7NJTJW34ME7R3UJIGUJW","json":"https://pith.science/pith/PTAQZQ7NJTJW34ME7R3UJIGUJW.json","graph_json":"https://pith.science/api/pith-number/PTAQZQ7NJTJW34ME7R3UJIGUJW/graph.json","events_json":"https://pith.science/api/pith-number/PTAQZQ7NJTJW34ME7R3UJIGUJW/events.json","paper":"https://pith.science/paper/PTAQZQ7N"},"agent_actions":{"view_html":"https://pith.science/pith/PTAQZQ7NJTJW34ME7R3UJIGUJW","download_json":"https://pith.science/pith/PTAQZQ7NJTJW34ME7R3UJIGUJW.json","view_paper":"https://pith.science/paper/PTAQZQ7N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.00645&json=true","fetch_graph":"https://pith.science/api/pith-number/PTAQZQ7NJTJW34ME7R3UJIGUJW/graph.json","fetch_events":"https://pith.science/api/pith-number/PTAQZQ7NJTJW34ME7R3UJIGUJW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PTAQZQ7NJTJW34ME7R3UJIGUJW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PTAQZQ7NJTJW34ME7R3UJIGUJW/action/storage_attestation","attest_author":"https://pith.science/pith/PTAQZQ7NJTJW34ME7R3UJIGUJW/action/author_attestation","sign_citation":"https://pith.science/pith/PTAQZQ7NJTJW34ME7R3UJIGUJW/action/citation_signature","submit_replication":"https://pith.science/pith/PTAQZQ7NJTJW34ME7R3UJIGUJW/action/replication_record"}},"created_at":"2026-07-05T09:56:08.616195+00:00","updated_at":"2026-07-05T09:56:08.616195+00:00"}