{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TABZE2JBT3WOGAI27M7AKCB3AT","short_pith_number":"pith:TABZE2JB","schema_version":"1.0","canonical_sha256":"98039269219eece3011afb3e05083b04da2b071f1263f68a26d4bc3ab4232c62","source":{"kind":"arxiv","id":"2412.15023","version":3},"attestation_state":"computed","paper":{"title":"FolAI: Synchronized Foley Sound Generation with Semantic and Temporal Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Christian Marinoni, Danilo Comminiello, Emilian Postolache, Joshua D. Reiss, Luca Cosmo, Marco Comunit\\`a, Riccardo Fosco Gramaccioni","submitted_at":"2024-12-19T16:37:19Z","abstract_excerpt":"Traditional sound design workflows rely on manual alignment of audio events to visual cues, as in Foley sound design, where everyday actions like footsteps or object interactions are recreated to match the on-screen motion. This process is time-consuming, difficult to scale, and lacks automation tools that preserve creative intent. Despite recent advances in vision-to-audio generation, producing temporally coherent and semantically controllable sound effects from video remains a major challenge. To address these limitations, we introduce FolAI, a two-stage generative framework that decouples t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.15023","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2024-12-19T16:37:19Z","cross_cats_sorted":["cs.CV","cs.LG","cs.MM","eess.AS"],"title_canon_sha256":"64987c334df5530e19180545d3c8bf2766fe2f28909d931db8650b8de35b1a77","abstract_canon_sha256":"149171cafa0b4c4f244ce850b801f9e4968df1d2fda1ed7f1a83077575c33425"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:58:21.709293Z","signature_b64":"KXWukQJTrVNjVQ7dYaBtXHVXyoA90Pq5eANCbSQ2oYw2oNmfsnC5LqX63JNPqXopWD83tiF5LAikBDJBRqYrAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"98039269219eece3011afb3e05083b04da2b071f1263f68a26d4bc3ab4232c62","last_reissued_at":"2026-07-05T10:58:21.708737Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:58:21.708737Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FolAI: Synchronized Foley Sound Generation with Semantic and Temporal Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Christian Marinoni, Danilo Comminiello, Emilian Postolache, Joshua D. Reiss, Luca Cosmo, Marco Comunit\\`a, Riccardo Fosco Gramaccioni","submitted_at":"2024-12-19T16:37:19Z","abstract_excerpt":"Traditional sound design workflows rely on manual alignment of audio events to visual cues, as in Foley sound design, where everyday actions like footsteps or object interactions are recreated to match the on-screen motion. This process is time-consuming, difficult to scale, and lacks automation tools that preserve creative intent. Despite recent advances in vision-to-audio generation, producing temporally coherent and semantically controllable sound effects from video remains a major challenge. To address these limitations, we introduce FolAI, a two-stage generative framework that decouples t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.15023","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.15023/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.15023","created_at":"2026-07-05T10:58:21.708797+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.15023v3","created_at":"2026-07-05T10:58:21.708797+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.15023","created_at":"2026-07-05T10:58:21.708797+00:00"},{"alias_kind":"pith_short_12","alias_value":"TABZE2JBT3WO","created_at":"2026-07-05T10:58:21.708797+00:00"},{"alias_kind":"pith_short_16","alias_value":"TABZE2JBT3WOGAI2","created_at":"2026-07-05T10:58:21.708797+00:00"},{"alias_kind":"pith_short_8","alias_value":"TABZE2JB","created_at":"2026-07-05T10:58:21.708797+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.24550","citing_title":"Training-Free Multimodal Guidance for Video to Audio Generation","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TABZE2JBT3WOGAI27M7AKCB3AT","json":"https://pith.science/pith/TABZE2JBT3WOGAI27M7AKCB3AT.json","graph_json":"https://pith.science/api/pith-number/TABZE2JBT3WOGAI27M7AKCB3AT/graph.json","events_json":"https://pith.science/api/pith-number/TABZE2JBT3WOGAI27M7AKCB3AT/events.json","paper":"https://pith.science/paper/TABZE2JB"},"agent_actions":{"view_html":"https://pith.science/pith/TABZE2JBT3WOGAI27M7AKCB3AT","download_json":"https://pith.science/pith/TABZE2JBT3WOGAI27M7AKCB3AT.json","view_paper":"https://pith.science/paper/TABZE2JB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.15023&json=true","fetch_graph":"https://pith.science/api/pith-number/TABZE2JBT3WOGAI27M7AKCB3AT/graph.json","fetch_events":"https://pith.science/api/pith-number/TABZE2JBT3WOGAI27M7AKCB3AT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TABZE2JBT3WOGAI27M7AKCB3AT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TABZE2JBT3WOGAI27M7AKCB3AT/action/storage_attestation","attest_author":"https://pith.science/pith/TABZE2JBT3WOGAI27M7AKCB3AT/action/author_attestation","sign_citation":"https://pith.science/pith/TABZE2JBT3WOGAI27M7AKCB3AT/action/citation_signature","submit_replication":"https://pith.science/pith/TABZE2JBT3WOGAI27M7AKCB3AT/action/replication_record"}},"created_at":"2026-07-05T10:58:21.708797+00:00","updated_at":"2026-07-05T10:58:21.708797+00:00"}