{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HFCLD5ZDEUTAYWU3Z764ICSXZW","short_pith_number":"pith:HFCLD5ZD","schema_version":"1.0","canonical_sha256":"3944b1f72325260c5a9bcffdc40a57cdab0eb4c600275efa77e0dba539bf3ab1","source":{"kind":"arxiv","id":"2312.04483","version":1},"attestation_state":"computed","paper":{"title":"Hierarchical Spatio-temporal Decoupling for Text-to-Video Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Changxin Gao, Jiayu Wang, Nong Sang, Shiwei Zhang, Xiang Wang, Yingya Zhang, Yujie Wei, Zhiwu Qing","submitted_at":"2023-12-07T17:59:07Z","abstract_excerpt":"Despite diffusion models having shown powerful abilities to generate photorealistic images, generating videos that are realistic and diverse still remains in its infancy. One of the key reasons is that current methods intertwine spatial content and temporal dynamics together, leading to a notably increased complexity of text-to-video generation (T2V). In this work, we propose HiGen, a diffusion model-based method that improves performance by decoupling the spatial and temporal factors of videos from two perspectives, i.e., structure level and content level. At the structure level, we decompose"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.04483","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-07T17:59:07Z","cross_cats_sorted":[],"title_canon_sha256":"26b4f4770f75bbd7aec21594c25dc97eaf583c645c571165738b2ffe296c3915","abstract_canon_sha256":"91c050d75a576b865765a5b799f6909f00595bb23c8f533ab62abec3973cb71b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:21:35.037634Z","signature_b64":"Zes9lgtJ6j46btrPmDju/63ZtlYF2TwvnJme4sLU6xC6OyDPoWe3+airya5nQCRmS5n55Jt1sUvCRwOkI7j1AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3944b1f72325260c5a9bcffdc40a57cdab0eb4c600275efa77e0dba539bf3ab1","last_reissued_at":"2026-07-05T07:21:35.037136Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:21:35.037136Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hierarchical Spatio-temporal Decoupling for Text-to-Video Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Changxin Gao, Jiayu Wang, Nong Sang, Shiwei Zhang, Xiang Wang, Yingya Zhang, Yujie Wei, Zhiwu Qing","submitted_at":"2023-12-07T17:59:07Z","abstract_excerpt":"Despite diffusion models having shown powerful abilities to generate photorealistic images, generating videos that are realistic and diverse still remains in its infancy. One of the key reasons is that current methods intertwine spatial content and temporal dynamics together, leading to a notably increased complexity of text-to-video generation (T2V). In this work, we propose HiGen, a diffusion model-based method that improves performance by decoupling the spatial and temporal factors of videos from two perspectives, i.e., structure level and content level. At the structure level, we decompose"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.04483","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.04483/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.04483","created_at":"2026-07-05T07:21:35.037197+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.04483v1","created_at":"2026-07-05T07:21:35.037197+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.04483","created_at":"2026-07-05T07:21:35.037197+00:00"},{"alias_kind":"pith_short_12","alias_value":"HFCLD5ZDEUTA","created_at":"2026-07-05T07:21:35.037197+00:00"},{"alias_kind":"pith_short_16","alias_value":"HFCLD5ZDEUTAYWU3","created_at":"2026-07-05T07:21:35.037197+00:00"},{"alias_kind":"pith_short_8","alias_value":"HFCLD5ZD","created_at":"2026-07-05T07:21:35.037197+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.23690","citing_title":"SynMotion: Semantic-Visual Adaptation for Motion Customized Video Generation","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HFCLD5ZDEUTAYWU3Z764ICSXZW","json":"https://pith.science/pith/HFCLD5ZDEUTAYWU3Z764ICSXZW.json","graph_json":"https://pith.science/api/pith-number/HFCLD5ZDEUTAYWU3Z764ICSXZW/graph.json","events_json":"https://pith.science/api/pith-number/HFCLD5ZDEUTAYWU3Z764ICSXZW/events.json","paper":"https://pith.science/paper/HFCLD5ZD"},"agent_actions":{"view_html":"https://pith.science/pith/HFCLD5ZDEUTAYWU3Z764ICSXZW","download_json":"https://pith.science/pith/HFCLD5ZDEUTAYWU3Z764ICSXZW.json","view_paper":"https://pith.science/paper/HFCLD5ZD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.04483&json=true","fetch_graph":"https://pith.science/api/pith-number/HFCLD5ZDEUTAYWU3Z764ICSXZW/graph.json","fetch_events":"https://pith.science/api/pith-number/HFCLD5ZDEUTAYWU3Z764ICSXZW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HFCLD5ZDEUTAYWU3Z764ICSXZW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HFCLD5ZDEUTAYWU3Z764ICSXZW/action/storage_attestation","attest_author":"https://pith.science/pith/HFCLD5ZDEUTAYWU3Z764ICSXZW/action/author_attestation","sign_citation":"https://pith.science/pith/HFCLD5ZDEUTAYWU3Z764ICSXZW/action/citation_signature","submit_replication":"https://pith.science/pith/HFCLD5ZDEUTAYWU3Z764ICSXZW/action/replication_record"}},"created_at":"2026-07-05T07:21:35.037197+00:00","updated_at":"2026-07-05T07:21:35.037197+00:00"}