{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:PH7ZJU7FYYVVMCVT7UQMG4ZJGZ","short_pith_number":"pith:PH7ZJU7F","schema_version":"1.0","canonical_sha256":"79ff94d3e5c62b560ab3fd20c373293658cf76a779d663e5e87d605fb4b0dd2b","source":{"kind":"arxiv","id":"2608.11913","version":1},"attestation_state":"computed","paper":{"title":"HarmoniDPO: Video-guided Audio Generation via Preference-Optimized Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Kaipeng Zhang, Wenshuo Peng","submitted_at":"2026-08-12T10:45:26Z","abstract_excerpt":"Video-to-audio (V2A) generation faces significant challenges in achieving precise temporal synchronization and high perceptual quality due to the complex, ambiguous relationship between visual and auditory cues. Existing methods typically compress video inputs into single feature representations, leading to significant loss of temporal dynamics and fine-grained visual information. These approaches also rely on reconstruction-based training objectives that poorly correlate with human perceptual judgments of audio quality and appropriateness. We propose HarmoniDPO, a novel framework that integra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.11913","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-08-12T10:45:26Z","cross_cats_sorted":[],"title_canon_sha256":"46ada26bd9c818254a9379eb28792c539351b6938bc492525e2925f067257e2b","abstract_canon_sha256":"22d2c8c6d03766c39f89b95be9cb7323550dc5a9f37c3bb5e6599655e0ba5b7c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-13T01:28:54.180895Z","signature_b64":"dDfc7OBK03If4tblkZgk3XFuJFZIXR96wMdJVgPHiDnhh9RT1nZsevkS8iHBn86tUgJc+0dt+010cZW/CPGICA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"79ff94d3e5c62b560ab3fd20c373293658cf76a779d663e5e87d605fb4b0dd2b","last_reissued_at":"2026-08-13T01:28:54.178388Z","signature_status":"signed_v1","first_computed_at":"2026-08-13T01:28:54.178388Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HarmoniDPO: Video-guided Audio Generation via Preference-Optimized Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Kaipeng Zhang, Wenshuo Peng","submitted_at":"2026-08-12T10:45:26Z","abstract_excerpt":"Video-to-audio (V2A) generation faces significant challenges in achieving precise temporal synchronization and high perceptual quality due to the complex, ambiguous relationship between visual and auditory cues. Existing methods typically compress video inputs into single feature representations, leading to significant loss of temporal dynamics and fine-grained visual information. These approaches also rely on reconstruction-based training objectives that poorly correlate with human perceptual judgments of audio quality and appropriateness. We propose HarmoniDPO, a novel framework that integra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.11913","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.11913/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.11913","created_at":"2026-08-13T01:28:54.179609+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.11913v1","created_at":"2026-08-13T01:28:54.179609+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.11913","created_at":"2026-08-13T01:28:54.179609+00:00"},{"alias_kind":"pith_short_12","alias_value":"PH7ZJU7FYYVV","created_at":"2026-08-13T01:28:54.179609+00:00"},{"alias_kind":"pith_short_16","alias_value":"PH7ZJU7FYYVVMCVT","created_at":"2026-08-13T01:28:54.179609+00:00"},{"alias_kind":"pith_short_8","alias_value":"PH7ZJU7F","created_at":"2026-08-13T01:28:54.179609+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PH7ZJU7FYYVVMCVT7UQMG4ZJGZ","json":"https://pith.science/pith/PH7ZJU7FYYVVMCVT7UQMG4ZJGZ.json","graph_json":"https://pith.science/api/pith-number/PH7ZJU7FYYVVMCVT7UQMG4ZJGZ/graph.json","events_json":"https://pith.science/api/pith-number/PH7ZJU7FYYVVMCVT7UQMG4ZJGZ/events.json","paper":"https://pith.science/paper/PH7ZJU7F"},"agent_actions":{"view_html":"https://pith.science/pith/PH7ZJU7FYYVVMCVT7UQMG4ZJGZ","download_json":"https://pith.science/pith/PH7ZJU7FYYVVMCVT7UQMG4ZJGZ.json","view_paper":"https://pith.science/paper/PH7ZJU7F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.11913&json=true","fetch_graph":"https://pith.science/api/pith-number/PH7ZJU7FYYVVMCVT7UQMG4ZJGZ/graph.json","fetch_events":"https://pith.science/api/pith-number/PH7ZJU7FYYVVMCVT7UQMG4ZJGZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PH7ZJU7FYYVVMCVT7UQMG4ZJGZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PH7ZJU7FYYVVMCVT7UQMG4ZJGZ/action/storage_attestation","attest_author":"https://pith.science/pith/PH7ZJU7FYYVVMCVT7UQMG4ZJGZ/action/author_attestation","sign_citation":"https://pith.science/pith/PH7ZJU7FYYVVMCVT7UQMG4ZJGZ/action/citation_signature","submit_replication":"https://pith.science/pith/PH7ZJU7FYYVVMCVT7UQMG4ZJGZ/action/replication_record"}},"created_at":"2026-08-13T01:28:54.179609+00:00","updated_at":"2026-08-13T01:28:54.179609+00:00"}