{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GJC7RESAUOSS3CMYCDDTITMDDA","short_pith_number":"pith:GJC7RESA","schema_version":"1.0","canonical_sha256":"3245f89240a3a52d899810c7344d83180272541f93bf664fabdc2ccd93e8432c","source":{"kind":"arxiv","id":"2411.17459","version":3},"attestation_state":"computed","paper":{"title":"WF-VAE: Enhancing Video VAE by Wavelet-Driven Energy Flow for Latent Video Diffusion Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bin Lin, Liuhan Chen, Li Yuan, Shenghai Yuan, Xinhua Cheng, Yang Ye, Zongjian Li","submitted_at":"2024-11-26T14:23:53Z","abstract_excerpt":"Video Variational Autoencoder (VAE) encodes videos into a low-dimensional latent space, becoming a key component of most Latent Video Diffusion Models (LVDMs) to reduce model training costs. However, as the resolution and duration of generated videos increase, the encoding cost of Video VAEs becomes a limiting bottleneck in training LVDMs. Moreover, the block-wise inference method adopted by most LVDMs can lead to discontinuities of latent space when processing long-duration videos. The key to addressing the computational bottleneck lies in decomposing videos into distinct components and effic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.17459","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-26T14:23:53Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fb1cd50ede3e4fde5352113cb757164036498d58bcb47546655d6d5a68febe33","abstract_canon_sha256":"bb81582cfe4c4d5aab0ef6a2d789c5e7df6f51e922004a862359757056a784e5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:33.891409Z","signature_b64":"UdeXowZTvranH8G+DsXVwNAv951eh+S+8CHboDTb3eWBR4ThFq1nOq88EPbU5qmC8IQ7U2FxWh/KZ+MY6JPPAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3245f89240a3a52d899810c7344d83180272541f93bf664fabdc2ccd93e8432c","last_reissued_at":"2026-07-05T10:47:33.890947Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:33.890947Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WF-VAE: Enhancing Video VAE by Wavelet-Driven Energy Flow for Latent Video Diffusion Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bin Lin, Liuhan Chen, Li Yuan, Shenghai Yuan, Xinhua Cheng, Yang Ye, Zongjian Li","submitted_at":"2024-11-26T14:23:53Z","abstract_excerpt":"Video Variational Autoencoder (VAE) encodes videos into a low-dimensional latent space, becoming a key component of most Latent Video Diffusion Models (LVDMs) to reduce model training costs. However, as the resolution and duration of generated videos increase, the encoding cost of Video VAEs becomes a limiting bottleneck in training LVDMs. Moreover, the block-wise inference method adopted by most LVDMs can lead to discontinuities of latent space when processing long-duration videos. The key to addressing the computational bottleneck lies in decomposing videos into distinct components and effic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.17459","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.17459/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.17459","created_at":"2026-07-05T10:47:33.891005+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.17459v3","created_at":"2026-07-05T10:47:33.891005+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.17459","created_at":"2026-07-05T10:47:33.891005+00:00"},{"alias_kind":"pith_short_12","alias_value":"GJC7RESAUOSS","created_at":"2026-07-05T10:47:33.891005+00:00"},{"alias_kind":"pith_short_16","alias_value":"GJC7RESAUOSS3CMY","created_at":"2026-07-05T10:47:33.891005+00:00"},{"alias_kind":"pith_short_8","alias_value":"GJC7RESA","created_at":"2026-07-05T10:47:33.891005+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31590","citing_title":"TunerDiT: Training-free Progressive Steering of Diffusion Transformer for Multi-Event Video Generation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2502.10248","citing_title":"Step-Video-T2V Technical Report: The Practice, Challenges, and Future of Video Foundation Model","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2504.13074","citing_title":"SkyReels-V2: Infinite-length Film Generative Model","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15911","citing_title":"Efficient Video Diffusion Models: Advancements and Challenges","ref_index":75,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GJC7RESAUOSS3CMYCDDTITMDDA","json":"https://pith.science/pith/GJC7RESAUOSS3CMYCDDTITMDDA.json","graph_json":"https://pith.science/api/pith-number/GJC7RESAUOSS3CMYCDDTITMDDA/graph.json","events_json":"https://pith.science/api/pith-number/GJC7RESAUOSS3CMYCDDTITMDDA/events.json","paper":"https://pith.science/paper/GJC7RESA"},"agent_actions":{"view_html":"https://pith.science/pith/GJC7RESAUOSS3CMYCDDTITMDDA","download_json":"https://pith.science/pith/GJC7RESAUOSS3CMYCDDTITMDDA.json","view_paper":"https://pith.science/paper/GJC7RESA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.17459&json=true","fetch_graph":"https://pith.science/api/pith-number/GJC7RESAUOSS3CMYCDDTITMDDA/graph.json","fetch_events":"https://pith.science/api/pith-number/GJC7RESAUOSS3CMYCDDTITMDDA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GJC7RESAUOSS3CMYCDDTITMDDA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GJC7RESAUOSS3CMYCDDTITMDDA/action/storage_attestation","attest_author":"https://pith.science/pith/GJC7RESAUOSS3CMYCDDTITMDDA/action/author_attestation","sign_citation":"https://pith.science/pith/GJC7RESAUOSS3CMYCDDTITMDDA/action/citation_signature","submit_replication":"https://pith.science/pith/GJC7RESAUOSS3CMYCDDTITMDDA/action/replication_record"}},"created_at":"2026-07-05T10:47:33.891005+00:00","updated_at":"2026-07-05T10:47:33.891005+00:00"}