{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XNN2I2L2RTGKOE6AECMPSFDWCR","short_pith_number":"pith:XNN2I2L2","schema_version":"1.0","canonical_sha256":"bb5ba4697a8ccca713c02098f91476147639580e3e60a1bc3e849947ed498da5","source":{"kind":"arxiv","id":"2409.01199","version":2},"attestation_state":"computed","paper":{"title":"OD-VAE: An Omni-dimensional Video Compressor for Improving Latent Video Diffusion Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["eess.IV"],"primary_cat":"cs.CV","authors_text":"Bin Lin, Bin Zhu, Liuhan Chen, Li Yuan, Qian Wang, Shenghai Yuan, Xing Zhou, Xinhua Cheng, Zongjian Li","submitted_at":"2024-09-02T12:20:42Z","abstract_excerpt":"Variational Autoencoder (VAE), compressing videos into latent representations, is a crucial preceding component of Latent Video Diffusion Models (LVDMs). With the same reconstruction quality, the more sufficient the VAE's compression for videos is, the more efficient the LVDMs are. However, most LVDMs utilize 2D image VAE, whose compression for videos is only in the spatial dimension and often ignored in the temporal dimension. How to conduct temporal compression for videos in a VAE to obtain more concise latent representations while promising accurate reconstruction is seldom explored. To fil"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.01199","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-09-02T12:20:42Z","cross_cats_sorted":["eess.IV"],"title_canon_sha256":"df2f95196cac512da462cecb6573ded5ed16f8a68856d5a1083d16d8d4674ac3","abstract_canon_sha256":"3fce5fd21fc50d653d746d4c36c6c80ceac964a648b18a12aa7f353e523d3b73"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:04:40.225860Z","signature_b64":"HyHY+Drerg1Gba6dhqUzaHOw6TP/i9U1AQkl9mFlR0ywC1hnxRax9ZCgxYgwv4Bhd5FVuJXzpuvbkoVpPh3aAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb5ba4697a8ccca713c02098f91476147639580e3e60a1bc3e849947ed498da5","last_reissued_at":"2026-07-05T09:04:40.225521Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:04:40.225521Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OD-VAE: An Omni-dimensional Video Compressor for Improving Latent Video Diffusion Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["eess.IV"],"primary_cat":"cs.CV","authors_text":"Bin Lin, Bin Zhu, Liuhan Chen, Li Yuan, Qian Wang, Shenghai Yuan, Xing Zhou, Xinhua Cheng, Zongjian Li","submitted_at":"2024-09-02T12:20:42Z","abstract_excerpt":"Variational Autoencoder (VAE), compressing videos into latent representations, is a crucial preceding component of Latent Video Diffusion Models (LVDMs). With the same reconstruction quality, the more sufficient the VAE's compression for videos is, the more efficient the LVDMs are. However, most LVDMs utilize 2D image VAE, whose compression for videos is only in the spatial dimension and often ignored in the temporal dimension. How to conduct temporal compression for videos in a VAE to obtain more concise latent representations while promising accurate reconstruction is seldom explored. To fil"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.01199","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.01199/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.01199","created_at":"2026-07-05T09:04:40.225575+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.01199v2","created_at":"2026-07-05T09:04:40.225575+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.01199","created_at":"2026-07-05T09:04:40.225575+00:00"},{"alias_kind":"pith_short_12","alias_value":"XNN2I2L2RTGK","created_at":"2026-07-05T09:04:40.225575+00:00"},{"alias_kind":"pith_short_16","alias_value":"XNN2I2L2RTGKOE6A","created_at":"2026-07-05T09:04:40.225575+00:00"},{"alias_kind":"pith_short_8","alias_value":"XNN2I2L2","created_at":"2026-07-05T09:04:40.225575+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17590","citing_title":"TivTok: Broadcasting Time-Invariant Tokens for Scalable Video Tokenization","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04410","citing_title":"Ultra-Fast Neural Video Compression","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2412.03603","citing_title":"HunyuanVideo: A Systematic Framework For Large Video Generative Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2603.17812","citing_title":"ChopGrad: Pixel-Wise Losses for Latent Video Diffusion via Truncated Backpropagation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02134","citing_title":"Video Generation with Predictive Latents","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07354","citing_title":"Task-Oriented Communication for Human Action Understanding via Edge-Cloud Co-Inference","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05171","citing_title":"Modality-Aware and Anatomical Vector-Quantized Autoencoding for Multimodal Brain MRI","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15911","citing_title":"Efficient Video Diffusion Models: Advancements and Challenges","ref_index":235,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XNN2I2L2RTGKOE6AECMPSFDWCR","json":"https://pith.science/pith/XNN2I2L2RTGKOE6AECMPSFDWCR.json","graph_json":"https://pith.science/api/pith-number/XNN2I2L2RTGKOE6AECMPSFDWCR/graph.json","events_json":"https://pith.science/api/pith-number/XNN2I2L2RTGKOE6AECMPSFDWCR/events.json","paper":"https://pith.science/paper/XNN2I2L2"},"agent_actions":{"view_html":"https://pith.science/pith/XNN2I2L2RTGKOE6AECMPSFDWCR","download_json":"https://pith.science/pith/XNN2I2L2RTGKOE6AECMPSFDWCR.json","view_paper":"https://pith.science/paper/XNN2I2L2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.01199&json=true","fetch_graph":"https://pith.science/api/pith-number/XNN2I2L2RTGKOE6AECMPSFDWCR/graph.json","fetch_events":"https://pith.science/api/pith-number/XNN2I2L2RTGKOE6AECMPSFDWCR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XNN2I2L2RTGKOE6AECMPSFDWCR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XNN2I2L2RTGKOE6AECMPSFDWCR/action/storage_attestation","attest_author":"https://pith.science/pith/XNN2I2L2RTGKOE6AECMPSFDWCR/action/author_attestation","sign_citation":"https://pith.science/pith/XNN2I2L2RTGKOE6AECMPSFDWCR/action/citation_signature","submit_replication":"https://pith.science/pith/XNN2I2L2RTGKOE6AECMPSFDWCR/action/replication_record"}},"created_at":"2026-07-05T09:04:40.225575+00:00","updated_at":"2026-07-05T09:04:40.225575+00:00"}