{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:K4GPZFZUFGRS2YQ2MEEGRIJY5M","short_pith_number":"pith:K4GPZFZU","schema_version":"1.0","canonical_sha256":"570cfc973429a32d621a610868a138eb0912508702465c6df4e7a6857135355b","source":{"kind":"arxiv","id":"2502.04507","version":3},"attestation_state":"computed","paper":{"title":"Fast Video Generation with Sliding Tile Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hangliang Ding, Hao Zhang, Ion Stoica, Peiyuan Zhang, Runlong Su, Yongqi Chen, Zhengzhong Liu","submitted_at":"2025-02-06T21:17:09Z","abstract_excerpt":"Diffusion Transformers (DiTs) with 3D full attention power state-of-the-art video generation, but suffer from prohibitive compute cost -- when generating just a 5-second 720P video, attention alone takes 800 out of 945 seconds of total inference time. This paper introduces sliding tile attention (STA) to address this challenge. STA leverages the observation that attention scores in pretrained video diffusion models predominantly concentrate within localized 3D windows. By sliding and attending over the local spatial-temporal region, STA eliminates redundancy from full attention. Unlike traditi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.04507","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-06T21:17:09Z","cross_cats_sorted":[],"title_canon_sha256":"2e1730b560bdc9fe70f86ab7624467436776ea6c0ba4434b4b9c1f3a4fab1732","abstract_canon_sha256":"f6900eb829f07d8de2faf8d6c8dc074067164974d5a6c1a24c89a7475e1a75a0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:05.924691Z","signature_b64":"9Kgrz9Skt9bsD2JzFIuJ7+Q1S/Lzw0c4F7t/BdyI1J1LJWoaBcRYXz1t1UPjOkUcpvtx3epc8hogvD1PfrPECg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"570cfc973429a32d621a610868a138eb0912508702465c6df4e7a6857135355b","last_reissued_at":"2026-07-05T11:16:05.923990Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:05.923990Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fast Video Generation with Sliding Tile Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hangliang Ding, Hao Zhang, Ion Stoica, Peiyuan Zhang, Runlong Su, Yongqi Chen, Zhengzhong Liu","submitted_at":"2025-02-06T21:17:09Z","abstract_excerpt":"Diffusion Transformers (DiTs) with 3D full attention power state-of-the-art video generation, but suffer from prohibitive compute cost -- when generating just a 5-second 720P video, attention alone takes 800 out of 945 seconds of total inference time. This paper introduces sliding tile attention (STA) to address this challenge. STA leverages the observation that attention scores in pretrained video diffusion models predominantly concentrate within localized 3D windows. By sliding and attending over the local spatial-temporal region, STA eliminates redundancy from full attention. Unlike traditi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.04507","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.04507/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.04507","created_at":"2026-07-05T11:16:05.924080+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.04507v3","created_at":"2026-07-05T11:16:05.924080+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.04507","created_at":"2026-07-05T11:16:05.924080+00:00"},{"alias_kind":"pith_short_12","alias_value":"K4GPZFZUFGRS","created_at":"2026-07-05T11:16:05.924080+00:00"},{"alias_kind":"pith_short_16","alias_value":"K4GPZFZUFGRS2YQ2","created_at":"2026-07-05T11:16:05.924080+00:00"},{"alias_kind":"pith_short_8","alias_value":"K4GPZFZU","created_at":"2026-07-05T11:16:05.924080+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06631","citing_title":"Dynamic-in-Few-Step: Unifying Dynamic Computation and Few-Step Distillation for Efficient Video Generation","ref_index":40,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12688","citing_title":"M*: A Modular, Extensible, Serving System for Multimodal Models","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31158","citing_title":"Light Interaction: Training-Free Inference Acceleration for Interactive Video World Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04569","citing_title":"LIVEditor-14B: Lightning Unified Video Editing via In-Context Sparse Attention","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30557","citing_title":"EcoVideo: Entropy-Orchestrated Video Generation Paradigm in Cloud-Edge Dynamics","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28691","citing_title":"OSP-Next: Efficient High-Quality Video Generation with Sparse Sequence Parallelism, HiF8 Quantization, and Reinforcement Learning","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31057","citing_title":"LVSA: Training-Free Sparse Attention for Long Video Diffusion","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23445","citing_title":"DFSAttn: Dynamic Fine-grained Sparse Attention for Efficient Video Generation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2603.21002","citing_title":"SURF: Signature-Retained Fast Video Generation","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2505.18875","citing_title":"Sparse VideoGen2: Accelerate Video Generation with Sparse Attention via Semantic-Aware Permutation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23980","citing_title":"Towards Redundancy Reduction in Diffusion Models for Efficient Video Super-Resolution","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09721","citing_title":"FrameDiT: Diffusion Transformer with Matrix Attention for Efficient Video Generation","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14513","citing_title":"HEART: Exploiting Head Heterogeneity in Sparse Attention for Video Diffusion","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14269","citing_title":"PhyMotion: Structured 3D Motion Reward for Physics-Grounded Human Video Generation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18870","citing_title":"HunyuanVideo 1.5 Technical Report","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20470","citing_title":"DynamicRad: Content-Adaptive Sparse Attention for Long Video Diffusion","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12219","citing_title":"Ride the Wave: Precision-Allocated Sparse Attention for Smooth Video Generation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10103","citing_title":"Long-Horizon Streaming Video Generation via Hybrid Attention with Decoupled Distillation","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18348","citing_title":"AdaCluster: Adaptive Query-Key Clustering for Sparse Attention in Video Generation","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K4GPZFZUFGRS2YQ2MEEGRIJY5M","json":"https://pith.science/pith/K4GPZFZUFGRS2YQ2MEEGRIJY5M.json","graph_json":"https://pith.science/api/pith-number/K4GPZFZUFGRS2YQ2MEEGRIJY5M/graph.json","events_json":"https://pith.science/api/pith-number/K4GPZFZUFGRS2YQ2MEEGRIJY5M/events.json","paper":"https://pith.science/paper/K4GPZFZU"},"agent_actions":{"view_html":"https://pith.science/pith/K4GPZFZUFGRS2YQ2MEEGRIJY5M","download_json":"https://pith.science/pith/K4GPZFZUFGRS2YQ2MEEGRIJY5M.json","view_paper":"https://pith.science/paper/K4GPZFZU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.04507&json=true","fetch_graph":"https://pith.science/api/pith-number/K4GPZFZUFGRS2YQ2MEEGRIJY5M/graph.json","fetch_events":"https://pith.science/api/pith-number/K4GPZFZUFGRS2YQ2MEEGRIJY5M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K4GPZFZUFGRS2YQ2MEEGRIJY5M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K4GPZFZUFGRS2YQ2MEEGRIJY5M/action/storage_attestation","attest_author":"https://pith.science/pith/K4GPZFZUFGRS2YQ2MEEGRIJY5M/action/author_attestation","sign_citation":"https://pith.science/pith/K4GPZFZUFGRS2YQ2MEEGRIJY5M/action/citation_signature","submit_replication":"https://pith.science/pith/K4GPZFZUFGRS2YQ2MEEGRIJY5M/action/replication_record"}},"created_at":"2026-07-05T11:16:05.924080+00:00","updated_at":"2026-07-05T11:16:05.924080+00:00"}