{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CUGEH4MEB5ZIFS4QE4A2KPPHDI","short_pith_number":"pith:CUGEH4ME","schema_version":"1.0","canonical_sha256":"150c43f1840f7282cb902701a53de71a0f8e4889dadf5208be50e2a9b0594846","source":{"kind":"arxiv","id":"2409.08601","version":2},"attestation_state":"computed","paper":{"title":"STA-V2A: Video-to-Audio Generation with Semantic and Temporal Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Chenxing Li, Dong Yu, Manjie Xu, Rilin Chen, Wei Liang, Yong Ren, Yu Gu","submitted_at":"2024-09-13T07:31:44Z","abstract_excerpt":"Visual and auditory perception are two crucial ways humans experience the world. Text-to-video generation has made remarkable progress over the past year, but the absence of harmonious audio in generated video limits its broader applications. In this paper, we propose Semantic and Temporal Aligned Video-to-Audio (STA-V2A), an approach that enhances audio generation from videos by extracting both local temporal and global semantic video features and combining these refined video features with text as cross-modal guidance. To address the issue of information redundancy in videos, we propose an o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.08601","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2024-09-13T07:31:44Z","cross_cats_sorted":["cs.MM","eess.AS"],"title_canon_sha256":"7f44ac90616ea28be2fd53f123506f18846241802cedd1817d6b866ef4b898f4","abstract_canon_sha256":"b84eac624627e9a28de6aacbd871a79b6f9058a2af83fc18bee3e8fdaf9fc6c8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:37:32.865427Z","signature_b64":"CTYQIhxGXQJKguN73C1CcgafYC85Y1mh5lh7lnfHLOK4PkLlDeBn1dfZx83JLK5ysABVBGUZJ5uAyaVePC7nDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"150c43f1840f7282cb902701a53de71a0f8e4889dadf5208be50e2a9b0594846","last_reissued_at":"2026-07-05T10:37:32.864857Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:37:32.864857Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"STA-V2A: Video-to-Audio Generation with Semantic and Temporal Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Chenxing Li, Dong Yu, Manjie Xu, Rilin Chen, Wei Liang, Yong Ren, Yu Gu","submitted_at":"2024-09-13T07:31:44Z","abstract_excerpt":"Visual and auditory perception are two crucial ways humans experience the world. Text-to-video generation has made remarkable progress over the past year, but the absence of harmonious audio in generated video limits its broader applications. In this paper, we propose Semantic and Temporal Aligned Video-to-Audio (STA-V2A), an approach that enhances audio generation from videos by extracting both local temporal and global semantic video features and combining these refined video features with text as cross-modal guidance. To address the issue of information redundancy in videos, we propose an o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.08601","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.08601/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.08601","created_at":"2026-07-05T10:37:32.864940+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.08601v2","created_at":"2026-07-05T10:37:32.864940+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.08601","created_at":"2026-07-05T10:37:32.864940+00:00"},{"alias_kind":"pith_short_12","alias_value":"CUGEH4MEB5ZI","created_at":"2026-07-05T10:37:32.864940+00:00"},{"alias_kind":"pith_short_16","alias_value":"CUGEH4MEB5ZIFS4Q","created_at":"2026-07-05T10:37:32.864940+00:00"},{"alias_kind":"pith_short_8","alias_value":"CUGEH4ME","created_at":"2026-07-05T10:37:32.864940+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26244","citing_title":"LongAV-Compass: Towards Unified Evaluation of Minute-Scale Audio-Visual Generation Across T2AV, I2AV, and V2AV","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CUGEH4MEB5ZIFS4QE4A2KPPHDI","json":"https://pith.science/pith/CUGEH4MEB5ZIFS4QE4A2KPPHDI.json","graph_json":"https://pith.science/api/pith-number/CUGEH4MEB5ZIFS4QE4A2KPPHDI/graph.json","events_json":"https://pith.science/api/pith-number/CUGEH4MEB5ZIFS4QE4A2KPPHDI/events.json","paper":"https://pith.science/paper/CUGEH4ME"},"agent_actions":{"view_html":"https://pith.science/pith/CUGEH4MEB5ZIFS4QE4A2KPPHDI","download_json":"https://pith.science/pith/CUGEH4MEB5ZIFS4QE4A2KPPHDI.json","view_paper":"https://pith.science/paper/CUGEH4ME","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.08601&json=true","fetch_graph":"https://pith.science/api/pith-number/CUGEH4MEB5ZIFS4QE4A2KPPHDI/graph.json","fetch_events":"https://pith.science/api/pith-number/CUGEH4MEB5ZIFS4QE4A2KPPHDI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CUGEH4MEB5ZIFS4QE4A2KPPHDI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CUGEH4MEB5ZIFS4QE4A2KPPHDI/action/storage_attestation","attest_author":"https://pith.science/pith/CUGEH4MEB5ZIFS4QE4A2KPPHDI/action/author_attestation","sign_citation":"https://pith.science/pith/CUGEH4MEB5ZIFS4QE4A2KPPHDI/action/citation_signature","submit_replication":"https://pith.science/pith/CUGEH4MEB5ZIFS4QE4A2KPPHDI/action/replication_record"}},"created_at":"2026-07-05T10:37:32.864940+00:00","updated_at":"2026-07-05T10:37:32.864940+00:00"}