{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7II7GEZL4RLM6GPYQHWA33MO6T","short_pith_number":"pith:7II7GEZL","schema_version":"1.0","canonical_sha256":"fa11f3132be456cf19f881ec0ded8ef4e75c108bcab4120d0e791fc9b9744cf0","source":{"kind":"arxiv","id":"2404.16581","version":1},"attestation_state":"computed","paper":{"title":"AudioScenic: Audio-Driven Video Scene Editing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jun Xiao, Kaixin Shen, Linchao Zhu, Ruijie Quan, Yi Yang","submitted_at":"2024-04-25T12:55:58Z","abstract_excerpt":"Audio-driven visual scene editing endeavors to manipulate the visual background while leaving the foreground content unchanged, according to the given audio signals. Unlike current efforts focusing primarily on image editing, audio-driven video scene editing has not been extensively addressed. In this paper, we introduce AudioScenic, an audio-driven framework designed for video scene editing. AudioScenic integrates audio semantics into the visual scene through a temporal-aware audio semantic injection process. As our focus is on background editing, we further introduce a SceneMasker module, wh"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.16581","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-04-25T12:55:58Z","cross_cats_sorted":[],"title_canon_sha256":"4c2da9dcba786b33a1f595aaabed64ff7e5dec38d22fa6f5af618bcd669ea90e","abstract_canon_sha256":"c26192cf5f848af24f0e650454f81f05a92de2eb7edbce30dc326bfb4819352e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:12:09.378584Z","signature_b64":"M6vD3dXcAThkTKrRKgbO9C5avB3NqfnT3mCkvw7QtjeIwDG908HtqCnvoJkzgKrHYcp5qLu+9kS+Oua7JAD7Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fa11f3132be456cf19f881ec0ded8ef4e75c108bcab4120d0e791fc9b9744cf0","last_reissued_at":"2026-07-05T08:12:09.378169Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:12:09.378169Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AudioScenic: Audio-Driven Video Scene Editing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jun Xiao, Kaixin Shen, Linchao Zhu, Ruijie Quan, Yi Yang","submitted_at":"2024-04-25T12:55:58Z","abstract_excerpt":"Audio-driven visual scene editing endeavors to manipulate the visual background while leaving the foreground content unchanged, according to the given audio signals. Unlike current efforts focusing primarily on image editing, audio-driven video scene editing has not been extensively addressed. In this paper, we introduce AudioScenic, an audio-driven framework designed for video scene editing. AudioScenic integrates audio semantics into the visual scene through a temporal-aware audio semantic injection process. As our focus is on background editing, we further introduce a SceneMasker module, wh"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.16581","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.16581/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.16581","created_at":"2026-07-05T08:12:09.378234+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.16581v1","created_at":"2026-07-05T08:12:09.378234+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.16581","created_at":"2026-07-05T08:12:09.378234+00:00"},{"alias_kind":"pith_short_12","alias_value":"7II7GEZL4RLM","created_at":"2026-07-05T08:12:09.378234+00:00"},{"alias_kind":"pith_short_16","alias_value":"7II7GEZL4RLM6GPY","created_at":"2026-07-05T08:12:09.378234+00:00"},{"alias_kind":"pith_short_8","alias_value":"7II7GEZL","created_at":"2026-07-05T08:12:09.378234+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.10571","citing_title":"AVI-Edit: Audio-sync Video Instance Editing with Granularity-Aware Mask Refiner","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7II7GEZL4RLM6GPYQHWA33MO6T","json":"https://pith.science/pith/7II7GEZL4RLM6GPYQHWA33MO6T.json","graph_json":"https://pith.science/api/pith-number/7II7GEZL4RLM6GPYQHWA33MO6T/graph.json","events_json":"https://pith.science/api/pith-number/7II7GEZL4RLM6GPYQHWA33MO6T/events.json","paper":"https://pith.science/paper/7II7GEZL"},"agent_actions":{"view_html":"https://pith.science/pith/7II7GEZL4RLM6GPYQHWA33MO6T","download_json":"https://pith.science/pith/7II7GEZL4RLM6GPYQHWA33MO6T.json","view_paper":"https://pith.science/paper/7II7GEZL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.16581&json=true","fetch_graph":"https://pith.science/api/pith-number/7II7GEZL4RLM6GPYQHWA33MO6T/graph.json","fetch_events":"https://pith.science/api/pith-number/7II7GEZL4RLM6GPYQHWA33MO6T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7II7GEZL4RLM6GPYQHWA33MO6T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7II7GEZL4RLM6GPYQHWA33MO6T/action/storage_attestation","attest_author":"https://pith.science/pith/7II7GEZL4RLM6GPYQHWA33MO6T/action/author_attestation","sign_citation":"https://pith.science/pith/7II7GEZL4RLM6GPYQHWA33MO6T/action/citation_signature","submit_replication":"https://pith.science/pith/7II7GEZL4RLM6GPYQHWA33MO6T/action/replication_record"}},"created_at":"2026-07-05T08:12:09.378234+00:00","updated_at":"2026-07-05T08:12:09.378234+00:00"}