{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TRAAY2B4VIE5X4UGO4QV4PWKO2","short_pith_number":"pith:TRAAY2B4","schema_version":"1.0","canonical_sha256":"9c400c683caa09dbf28677215e3eca76a6db124bfd9ad4d7005d21ac68ce97da","source":{"kind":"arxiv","id":"2407.10387","version":1},"attestation_state":"computed","paper":{"title":"Masked Generative Video-to-Audio Transformers with Enhanced Synchronicity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","eess.AS"],"primary_cat":"cs.SD","authors_text":"Chunghsin Yeh, Ioannis Tsiamas, Joan Serr\\`a, Santiago Pascual","submitted_at":"2024-07-15T01:49:59Z","abstract_excerpt":"Video-to-audio (V2A) generation leverages visual-only video features to render plausible sounds that match the scene. Importantly, the generated sound onsets should match the visual actions that are aligned with them, otherwise unnatural synchronization artifacts arise. Recent works have explored the progression of conditioning sound generators on still images and then video features, focusing on quality and semantic matching while ignoring synchronization, or by sacrificing some amount of quality to focus on improving synchronization only. In this work, we propose a V2A generative model, name"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.10387","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2024-07-15T01:49:59Z","cross_cats_sorted":["cs.AI","cs.CV","eess.AS"],"title_canon_sha256":"226f7d5d0bf1832151dd501f9b7ac38af143b7bc5cea9187c227a558fcfefaec","abstract_canon_sha256":"84fd287e3a58e8b7ee5724544dda19df8d68fdc7e6bd9ccb64f73f5f67309282"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:43:58.941805Z","signature_b64":"gpdP0ZD0ogAEP8xS3OwUONY6V8L1DYH+bkoQ09wav8NmjIpnrO+EpmgzG7KiZEqreM/H5TUz9NUeXncGT34jCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9c400c683caa09dbf28677215e3eca76a6db124bfd9ad4d7005d21ac68ce97da","last_reissued_at":"2026-07-05T08:43:58.941390Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:43:58.941390Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Masked Generative Video-to-Audio Transformers with Enhanced Synchronicity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","eess.AS"],"primary_cat":"cs.SD","authors_text":"Chunghsin Yeh, Ioannis Tsiamas, Joan Serr\\`a, Santiago Pascual","submitted_at":"2024-07-15T01:49:59Z","abstract_excerpt":"Video-to-audio (V2A) generation leverages visual-only video features to render plausible sounds that match the scene. Importantly, the generated sound onsets should match the visual actions that are aligned with them, otherwise unnatural synchronization artifacts arise. Recent works have explored the progression of conditioning sound generators on still images and then video features, focusing on quality and semantic matching while ignoring synchronization, or by sacrificing some amount of quality to focus on improving synchronization only. In this work, we propose a V2A generative model, name"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.10387","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.10387/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.10387","created_at":"2026-07-05T08:43:58.941447+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.10387v1","created_at":"2026-07-05T08:43:58.941447+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.10387","created_at":"2026-07-05T08:43:58.941447+00:00"},{"alias_kind":"pith_short_12","alias_value":"TRAAY2B4VIE5","created_at":"2026-07-05T08:43:58.941447+00:00"},{"alias_kind":"pith_short_16","alias_value":"TRAAY2B4VIE5X4UG","created_at":"2026-07-05T08:43:58.941447+00:00"},{"alias_kind":"pith_short_8","alias_value":"TRAAY2B4","created_at":"2026-07-05T08:43:58.941447+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.12136","citing_title":"Room Impulse Response Generation Conditioned on Acoustic Parameters","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TRAAY2B4VIE5X4UGO4QV4PWKO2","json":"https://pith.science/pith/TRAAY2B4VIE5X4UGO4QV4PWKO2.json","graph_json":"https://pith.science/api/pith-number/TRAAY2B4VIE5X4UGO4QV4PWKO2/graph.json","events_json":"https://pith.science/api/pith-number/TRAAY2B4VIE5X4UGO4QV4PWKO2/events.json","paper":"https://pith.science/paper/TRAAY2B4"},"agent_actions":{"view_html":"https://pith.science/pith/TRAAY2B4VIE5X4UGO4QV4PWKO2","download_json":"https://pith.science/pith/TRAAY2B4VIE5X4UGO4QV4PWKO2.json","view_paper":"https://pith.science/paper/TRAAY2B4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.10387&json=true","fetch_graph":"https://pith.science/api/pith-number/TRAAY2B4VIE5X4UGO4QV4PWKO2/graph.json","fetch_events":"https://pith.science/api/pith-number/TRAAY2B4VIE5X4UGO4QV4PWKO2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TRAAY2B4VIE5X4UGO4QV4PWKO2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TRAAY2B4VIE5X4UGO4QV4PWKO2/action/storage_attestation","attest_author":"https://pith.science/pith/TRAAY2B4VIE5X4UGO4QV4PWKO2/action/author_attestation","sign_citation":"https://pith.science/pith/TRAAY2B4VIE5X4UGO4QV4PWKO2/action/citation_signature","submit_replication":"https://pith.science/pith/TRAAY2B4VIE5X4UGO4QV4PWKO2/action/replication_record"}},"created_at":"2026-07-05T08:43:58.941447+00:00","updated_at":"2026-07-05T08:43:58.941447+00:00"}