{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HWI7UAYCKNPVJABKJ7IFIGE6UX","short_pith_number":"pith:HWI7UAYC","schema_version":"1.0","canonical_sha256":"3d91fa0302535f54802a4fd054189ea5f4743c06945be5c9b0cf17d6f8ae5b7e","source":{"kind":"arxiv","id":"2505.14687","version":1},"attestation_state":"computed","paper":{"title":"Grouping First, Attending Smartly: Training-Free Acceleration for Diffusion Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alan Yuille, Ju He, Liang-Chieh Chen, Qihang Yu, Sucheng Ren","submitted_at":"2025-05-20T17:59:59Z","abstract_excerpt":"Diffusion-based Transformers have demonstrated impressive generative capabilities, but their high computational costs hinder practical deployment, for example, generating an $8192\\times 8192$ image can take over an hour on an A100 GPU. In this work, we propose GRAT (\\textbf{GR}ouping first, \\textbf{AT}tending smartly), a training-free attention acceleration strategy for fast image and video generation without compromising output quality. The key insight is to exploit the inherent sparsity in learned attention maps (which tend to be locally focused) in pretrained Diffusion Transformers and leve"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.14687","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-20T17:59:59Z","cross_cats_sorted":[],"title_canon_sha256":"3a3d0db060b58468897936be0ee2a2f18e16a563c3960562b953f0f792eddc0a","abstract_canon_sha256":"babf70aace65b6d8ec648e726a0bb407f9a1599f0093d5248ad0c0e5ba1a8def"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:17.653842Z","signature_b64":"fLeSH/wFibmvFwnTG3Kv7bta89C7lla41Aa7nflqoCOTgvkA9327828utOwvDY+Kdl2EWuhxGHLy7999v/F4Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3d91fa0302535f54802a4fd054189ea5f4743c06945be5c9b0cf17d6f8ae5b7e","last_reissued_at":"2026-07-05T11:06:17.653331Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:17.653331Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Grouping First, Attending Smartly: Training-Free Acceleration for Diffusion Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alan Yuille, Ju He, Liang-Chieh Chen, Qihang Yu, Sucheng Ren","submitted_at":"2025-05-20T17:59:59Z","abstract_excerpt":"Diffusion-based Transformers have demonstrated impressive generative capabilities, but their high computational costs hinder practical deployment, for example, generating an $8192\\times 8192$ image can take over an hour on an A100 GPU. In this work, we propose GRAT (\\textbf{GR}ouping first, \\textbf{AT}tending smartly), a training-free attention acceleration strategy for fast image and video generation without compromising output quality. The key insight is to exploit the inherent sparsity in learned attention maps (which tend to be locally focused) in pretrained Diffusion Transformers and leve"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.14687","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.14687/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.14687","created_at":"2026-07-05T11:06:17.653400+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.14687v1","created_at":"2026-07-05T11:06:17.653400+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.14687","created_at":"2026-07-05T11:06:17.653400+00:00"},{"alias_kind":"pith_short_12","alias_value":"HWI7UAYCKNPV","created_at":"2026-07-05T11:06:17.653400+00:00"},{"alias_kind":"pith_short_16","alias_value":"HWI7UAYCKNPVJABK","created_at":"2026-07-05T11:06:17.653400+00:00"},{"alias_kind":"pith_short_8","alias_value":"HWI7UAYC","created_at":"2026-07-05T11:06:17.653400+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.09313","citing_title":"Attention Sinks in Diffusion Transformers: A Causal Analysis","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2603.06351","citing_title":"DC-DiT: Adaptive Compute and Elastic Inference for Visual Generation via Dynamic Chunking","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09313","citing_title":"Attention Sinks in Diffusion Transformers: A Causal Analysis","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09313","citing_title":"Attention Sinks in Diffusion Transformers: A Causal Analysis","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04913","citing_title":"A Frame is Worth One Token: Efficient Generative World Modeling with Delta Tokens","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15521","citing_title":"Frequency-Aware Flow Matching for High-Quality Image Generation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15911","citing_title":"Efficient Video Diffusion Models: Advancements and Challenges","ref_index":110,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HWI7UAYCKNPVJABKJ7IFIGE6UX","json":"https://pith.science/pith/HWI7UAYCKNPVJABKJ7IFIGE6UX.json","graph_json":"https://pith.science/api/pith-number/HWI7UAYCKNPVJABKJ7IFIGE6UX/graph.json","events_json":"https://pith.science/api/pith-number/HWI7UAYCKNPVJABKJ7IFIGE6UX/events.json","paper":"https://pith.science/paper/HWI7UAYC"},"agent_actions":{"view_html":"https://pith.science/pith/HWI7UAYCKNPVJABKJ7IFIGE6UX","download_json":"https://pith.science/pith/HWI7UAYCKNPVJABKJ7IFIGE6UX.json","view_paper":"https://pith.science/paper/HWI7UAYC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.14687&json=true","fetch_graph":"https://pith.science/api/pith-number/HWI7UAYCKNPVJABKJ7IFIGE6UX/graph.json","fetch_events":"https://pith.science/api/pith-number/HWI7UAYCKNPVJABKJ7IFIGE6UX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HWI7UAYCKNPVJABKJ7IFIGE6UX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HWI7UAYCKNPVJABKJ7IFIGE6UX/action/storage_attestation","attest_author":"https://pith.science/pith/HWI7UAYCKNPVJABKJ7IFIGE6UX/action/author_attestation","sign_citation":"https://pith.science/pith/HWI7UAYCKNPVJABKJ7IFIGE6UX/action/citation_signature","submit_replication":"https://pith.science/pith/HWI7UAYCKNPVJABKJ7IFIGE6UX/action/replication_record"}},"created_at":"2026-07-05T11:06:17.653400+00:00","updated_at":"2026-07-05T11:06:17.653400+00:00"}