{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SGQ2PBMGXDWNFH35VGOZMC7DYA","short_pith_number":"pith:SGQ2PBMG","schema_version":"1.0","canonical_sha256":"91a1a78586b8ecd29f7da99d960be3c035f3a80f1ed5df45479284dc94aaa159","source":{"kind":"arxiv","id":"2408.12588","version":3},"attestation_state":"computed","paper":{"title":"Real-Time Video Generation with Pyramid Attention Broadcast","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.CV","authors_text":"Kai Wang, Xiaolong Jin, Xuanlei Zhao, Yang You","submitted_at":"2024-08-22T17:54:21Z","abstract_excerpt":"We present Pyramid Attention Broadcast (PAB), a real-time, high quality and training-free approach for DiT-based video generation. Our method is founded on the observation that attention difference in the diffusion process exhibits a U-shaped pattern, indicating significant redundancy. We mitigate this by broadcasting attention outputs to subsequent steps in a pyramid style. It applies different broadcast strategies to each attention based on their variance for best efficiency. We further introduce broadcast sequence parallel for more efficient distributed inference. PAB demonstrates up to 10."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.12588","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-08-22T17:54:21Z","cross_cats_sorted":["cs.DC"],"title_canon_sha256":"d3809e165172940a372825a96a13e2760a97564bd749a7d767ae1e62210c2f74","abstract_canon_sha256":"4e3c32416dbae0da2b93f797f01df417c9837f43542a20705f34871ba0652883"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:20:39.495373Z","signature_b64":"38+w+5/SHubPc12wIxeAjY88bk6lMaRt1ens5IkjNmAiJ8JiYybcnVit3sF4iArU0nQBt5tq84+Utc79FBz0CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"91a1a78586b8ecd29f7da99d960be3c035f3a80f1ed5df45479284dc94aaa159","last_reissued_at":"2026-07-05T10:20:39.494832Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:20:39.494832Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Real-Time Video Generation with Pyramid Attention Broadcast","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.CV","authors_text":"Kai Wang, Xiaolong Jin, Xuanlei Zhao, Yang You","submitted_at":"2024-08-22T17:54:21Z","abstract_excerpt":"We present Pyramid Attention Broadcast (PAB), a real-time, high quality and training-free approach for DiT-based video generation. Our method is founded on the observation that attention difference in the diffusion process exhibits a U-shaped pattern, indicating significant redundancy. We mitigate this by broadcasting attention outputs to subsequent steps in a pyramid style. It applies different broadcast strategies to each attention based on their variance for best efficiency. We further introduce broadcast sequence parallel for more efficient distributed inference. PAB demonstrates up to 10."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.12588","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.12588/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.12588","created_at":"2026-07-05T10:20:39.494898+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.12588v3","created_at":"2026-07-05T10:20:39.494898+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.12588","created_at":"2026-07-05T10:20:39.494898+00:00"},{"alias_kind":"pith_short_12","alias_value":"SGQ2PBMGXDWN","created_at":"2026-07-05T10:20:39.494898+00:00"},{"alias_kind":"pith_short_16","alias_value":"SGQ2PBMGXDWNFH35","created_at":"2026-07-05T10:20:39.494898+00:00"},{"alias_kind":"pith_short_8","alias_value":"SGQ2PBMG","created_at":"2026-07-05T10:20:39.494898+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23743","citing_title":"Sol Video Inference Engine: Agent-Native Full-Stack Acceleration Framework for Efficient Video Generation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06309","citing_title":"RhymeFlow: Training-Free Acceleration for Video Generation with Asynchronous Denoising Flow Scheduling","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31158","citing_title":"Light Interaction: Training-Free Inference Acceleration for Interactive Video World Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31026","citing_title":"OTCache: Optimal Transport for Geometry-Aware Caching in Diffusion Models","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02586","citing_title":"Fewer, Better Frames: A Compute-Normalized Proof of Concept for Coherence-First World-Model Rendering with Model-Guided FSR4 Frame Generation","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27336","citing_title":"PARE: Pruning and Adaptive Routing for Efficient Video Generation","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23381","citing_title":"VDE: Training-Free Accelerating Rectified Flow Model via Velocity Decomposition and Estimation","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2405.14430","citing_title":"PipeFusion: Patch-level Pipeline Parallelism for Diffusion Transformers Inference","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2412.03603","citing_title":"HunyuanVideo: A Systematic Framework For Large Video Generative Models","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22015","citing_title":"ORBIS: Output-Guided Token Reduction with Distribution-Aware Matching for Video Diffusion Acceleration","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2602.05449","citing_title":"DisCa: Accelerating Video Diffusion Transformers with Distillation-Compatible Learnable Feature Caching","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01725","citing_title":"Motion-Aware Caching for Efficient Autoregressive Video Generation","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24527","citing_title":"Training Agents Inside of Scalable World Models","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02979","citing_title":"Not All Frames Deserve Full Computation: Accelerating Autoregressive Video Generation via Selective Computation and Predictive Extrapolation","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24447","citing_title":"Characterizing Vision-Language-Action Models across XPUs: Constraints and Acceleration for On-Robot Deployment","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2410.13720","citing_title":"Movie Gen: A Cast of Media Foundation Models","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20470","citing_title":"DynamicRad: Content-Adaptive Sparse Attention for Long Video Diffusion","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16492","citing_title":"LayerCache: Exploiting Layer-wise Velocity Heterogeneity for Efficient Flow Matching Inference","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01725","citing_title":"Motion-Aware Caching for Efficient Autoregressive Video Generation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18348","citing_title":"AdaCluster: Adaptive Query-Key Clustering for Sparse Attention in Video Generation","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15911","citing_title":"Efficient Video Diffusion Models: Advancements and Challenges","ref_index":195,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SGQ2PBMGXDWNFH35VGOZMC7DYA","json":"https://pith.science/pith/SGQ2PBMGXDWNFH35VGOZMC7DYA.json","graph_json":"https://pith.science/api/pith-number/SGQ2PBMGXDWNFH35VGOZMC7DYA/graph.json","events_json":"https://pith.science/api/pith-number/SGQ2PBMGXDWNFH35VGOZMC7DYA/events.json","paper":"https://pith.science/paper/SGQ2PBMG"},"agent_actions":{"view_html":"https://pith.science/pith/SGQ2PBMGXDWNFH35VGOZMC7DYA","download_json":"https://pith.science/pith/SGQ2PBMGXDWNFH35VGOZMC7DYA.json","view_paper":"https://pith.science/paper/SGQ2PBMG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.12588&json=true","fetch_graph":"https://pith.science/api/pith-number/SGQ2PBMGXDWNFH35VGOZMC7DYA/graph.json","fetch_events":"https://pith.science/api/pith-number/SGQ2PBMGXDWNFH35VGOZMC7DYA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SGQ2PBMGXDWNFH35VGOZMC7DYA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SGQ2PBMGXDWNFH35VGOZMC7DYA/action/storage_attestation","attest_author":"https://pith.science/pith/SGQ2PBMGXDWNFH35VGOZMC7DYA/action/author_attestation","sign_citation":"https://pith.science/pith/SGQ2PBMGXDWNFH35VGOZMC7DYA/action/citation_signature","submit_replication":"https://pith.science/pith/SGQ2PBMGXDWNFH35VGOZMC7DYA/action/replication_record"}},"created_at":"2026-07-05T10:20:39.494898+00:00","updated_at":"2026-07-05T10:20:39.494898+00:00"}