{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:QCP5EQSPWWDFCJLJEDYCKWUNII","short_pith_number":"pith:QCP5EQSP","schema_version":"1.0","canonical_sha256":"809fd2424fb58651256920f0255a8d420278c6631bdcee461bfab961cd3d53e9","source":{"kind":"arxiv","id":"2608.02474","version":1},"attestation_state":"computed","paper":{"title":"EchoCache: Energy-Guided Cross-Modal Caching for Efficient Audio-Driven Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guojie Luo, Hailong Zou, Jiayu Chen, Maoliang Li, Rongshan Gao, Xiang Chen, Xiaoyu Wu, Xinhao Sun, Zihao Zheng","submitted_at":"2026-08-03T16:42:54Z","abstract_excerpt":"Audio-driven video generation (A2V) has achieved promising progress in synthesizing temporally coherent and audio-visually aligned videos, yet its inference remains expensive due to the iterative denoising process of diffusion models. Existing caching methods mainly exploit temporal redundancy in visual features while overlooking the cross-modal alignment of A2V, where audio drives visual generation with highly non-uniform temporal importance. In this paper, we identify two levels of misalignment in existing A2V caching methods: temporal-semantic and computation-storage misalignment. To addres"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.02474","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-08-03T16:42:54Z","cross_cats_sorted":[],"title_canon_sha256":"bbb192f064457a9bc9125e04a82d7068b7d21b87df9636569cff64493c488c12","abstract_canon_sha256":"0584662e53bddaad274e40d1c3de4d2904f7555284613a286b5c3f9456bb308e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-04T02:43:45.301234Z","signature_b64":"mw24CDZamn3/MIzBuNQ/BMAKI4kFoXGrFwE56+34VjhHtQeZrIeLhalWzZ2T8iGlppjFK2CbUIXgyoB3HBjEBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"809fd2424fb58651256920f0255a8d420278c6631bdcee461bfab961cd3d53e9","last_reissued_at":"2026-08-04T02:43:45.298988Z","signature_status":"signed_v1","first_computed_at":"2026-08-04T02:43:45.298988Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EchoCache: Energy-Guided Cross-Modal Caching for Efficient Audio-Driven Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guojie Luo, Hailong Zou, Jiayu Chen, Maoliang Li, Rongshan Gao, Xiang Chen, Xiaoyu Wu, Xinhao Sun, Zihao Zheng","submitted_at":"2026-08-03T16:42:54Z","abstract_excerpt":"Audio-driven video generation (A2V) has achieved promising progress in synthesizing temporally coherent and audio-visually aligned videos, yet its inference remains expensive due to the iterative denoising process of diffusion models. Existing caching methods mainly exploit temporal redundancy in visual features while overlooking the cross-modal alignment of A2V, where audio drives visual generation with highly non-uniform temporal importance. In this paper, we identify two levels of misalignment in existing A2V caching methods: temporal-semantic and computation-storage misalignment. To addres"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.02474","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.02474/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.02474","created_at":"2026-08-04T02:43:45.301262+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.02474v1","created_at":"2026-08-04T02:43:45.301262+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.02474","created_at":"2026-08-04T02:43:45.301262+00:00"},{"alias_kind":"pith_short_12","alias_value":"QCP5EQSPWWDF","created_at":"2026-08-04T02:43:45.301262+00:00"},{"alias_kind":"pith_short_16","alias_value":"QCP5EQSPWWDFCJLJ","created_at":"2026-08-04T02:43:45.301262+00:00"},{"alias_kind":"pith_short_8","alias_value":"QCP5EQSP","created_at":"2026-08-04T02:43:45.301262+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QCP5EQSPWWDFCJLJEDYCKWUNII","json":"https://pith.science/pith/QCP5EQSPWWDFCJLJEDYCKWUNII.json","graph_json":"https://pith.science/api/pith-number/QCP5EQSPWWDFCJLJEDYCKWUNII/graph.json","events_json":"https://pith.science/api/pith-number/QCP5EQSPWWDFCJLJEDYCKWUNII/events.json","paper":"https://pith.science/paper/QCP5EQSP"},"agent_actions":{"view_html":"https://pith.science/pith/QCP5EQSPWWDFCJLJEDYCKWUNII","download_json":"https://pith.science/pith/QCP5EQSPWWDFCJLJEDYCKWUNII.json","view_paper":"https://pith.science/paper/QCP5EQSP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.02474&json=true","fetch_graph":"https://pith.science/api/pith-number/QCP5EQSPWWDFCJLJEDYCKWUNII/graph.json","fetch_events":"https://pith.science/api/pith-number/QCP5EQSPWWDFCJLJEDYCKWUNII/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QCP5EQSPWWDFCJLJEDYCKWUNII/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QCP5EQSPWWDFCJLJEDYCKWUNII/action/storage_attestation","attest_author":"https://pith.science/pith/QCP5EQSPWWDFCJLJEDYCKWUNII/action/author_attestation","sign_citation":"https://pith.science/pith/QCP5EQSPWWDFCJLJEDYCKWUNII/action/citation_signature","submit_replication":"https://pith.science/pith/QCP5EQSPWWDFCJLJEDYCKWUNII/action/replication_record"}},"created_at":"2026-08-04T02:43:45.301262+00:00","updated_at":"2026-08-04T02:43:45.301262+00:00"}