{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IO6B5PMDC43KEUEWZR6JZL2PCQ","short_pith_number":"pith:IO6B5PMD","schema_version":"1.0","canonical_sha256":"43bc1ebd831736a25096cc7c9caf4f1420a60d65a82c900f011b0af962c0fd09","source":{"kind":"arxiv","id":"2504.12027","version":2},"attestation_state":"computed","paper":{"title":"Understanding Attention Mechanism in Video Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyan Liu, Chengyu Wang, Huan Ten, Jun Huang, Kailing Guo, Kui Jia, Tongtong Su","submitted_at":"2025-04-16T12:37:08Z","abstract_excerpt":"Text-to-video (T2V) synthesis models, such as OpenAI's Sora, have garnered significant attention due to their ability to generate high-quality videos from a text prompt. In diffusion-based T2V models, the attention mechanism is a critical component. However, it remains unclear what intermediate features are learned and how attention blocks in T2V models affect various aspects of video synthesis, such as image quality and temporal consistency. In this paper, we conduct an in-depth perturbation analysis of the spatial and temporal attention blocks of T2V models using an information-theoretic app"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.12027","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-04-16T12:37:08Z","cross_cats_sorted":[],"title_canon_sha256":"cd7cb8d20b01070527efa3f820229bcbdf6fa893760b5350827bde3072d897b0","abstract_canon_sha256":"cd74a335102adadb899ace375f83d53e693f317d15c3f96ec5f46dbd54296ccc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:50:13.411718Z","signature_b64":"XAz5lAMpJUmxfKt0vinrBARta+kecYiCILvgT11His73THL6g0GTRNG/cj5tBPF4cgLWcQnnkfoJM4pGMPMGAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"43bc1ebd831736a25096cc7c9caf4f1420a60d65a82c900f011b0af962c0fd09","last_reissued_at":"2026-07-05T10:50:13.411171Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:50:13.411171Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Attention Mechanism in Video Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyan Liu, Chengyu Wang, Huan Ten, Jun Huang, Kailing Guo, Kui Jia, Tongtong Su","submitted_at":"2025-04-16T12:37:08Z","abstract_excerpt":"Text-to-video (T2V) synthesis models, such as OpenAI's Sora, have garnered significant attention due to their ability to generate high-quality videos from a text prompt. In diffusion-based T2V models, the attention mechanism is a critical component. However, it remains unclear what intermediate features are learned and how attention blocks in T2V models affect various aspects of video synthesis, such as image quality and temporal consistency. In this paper, we conduct an in-depth perturbation analysis of the spatial and temporal attention blocks of T2V models using an information-theoretic app"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.12027","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.12027/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.12027","created_at":"2026-07-05T10:50:13.411235+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.12027v2","created_at":"2026-07-05T10:50:13.411235+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.12027","created_at":"2026-07-05T10:50:13.411235+00:00"},{"alias_kind":"pith_short_12","alias_value":"IO6B5PMDC43K","created_at":"2026-07-05T10:50:13.411235+00:00"},{"alias_kind":"pith_short_16","alias_value":"IO6B5PMDC43KEUEW","created_at":"2026-07-05T10:50:13.411235+00:00"},{"alias_kind":"pith_short_8","alias_value":"IO6B5PMD","created_at":"2026-07-05T10:50:13.411235+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.21901","citing_title":"CLEAR: Context-Aware Learning with End-to-End Mask-Free Inference for Adaptive Video Subtitle Removal","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IO6B5PMDC43KEUEWZR6JZL2PCQ","json":"https://pith.science/pith/IO6B5PMDC43KEUEWZR6JZL2PCQ.json","graph_json":"https://pith.science/api/pith-number/IO6B5PMDC43KEUEWZR6JZL2PCQ/graph.json","events_json":"https://pith.science/api/pith-number/IO6B5PMDC43KEUEWZR6JZL2PCQ/events.json","paper":"https://pith.science/paper/IO6B5PMD"},"agent_actions":{"view_html":"https://pith.science/pith/IO6B5PMDC43KEUEWZR6JZL2PCQ","download_json":"https://pith.science/pith/IO6B5PMDC43KEUEWZR6JZL2PCQ.json","view_paper":"https://pith.science/paper/IO6B5PMD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.12027&json=true","fetch_graph":"https://pith.science/api/pith-number/IO6B5PMDC43KEUEWZR6JZL2PCQ/graph.json","fetch_events":"https://pith.science/api/pith-number/IO6B5PMDC43KEUEWZR6JZL2PCQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IO6B5PMDC43KEUEWZR6JZL2PCQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IO6B5PMDC43KEUEWZR6JZL2PCQ/action/storage_attestation","attest_author":"https://pith.science/pith/IO6B5PMDC43KEUEWZR6JZL2PCQ/action/author_attestation","sign_citation":"https://pith.science/pith/IO6B5PMDC43KEUEWZR6JZL2PCQ/action/citation_signature","submit_replication":"https://pith.science/pith/IO6B5PMDC43KEUEWZR6JZL2PCQ/action/replication_record"}},"created_at":"2026-07-05T10:50:13.411235+00:00","updated_at":"2026-07-05T10:50:13.411235+00:00"}