{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UBFDTAJYHX42ELTASFBVPHXULZ","short_pith_number":"pith:UBFDTAJY","schema_version":"1.0","canonical_sha256":"a04a3981383df9a22e609143579ef45e64613df0f4358de9bf04a47b3f93b77d","source":{"kind":"arxiv","id":"2409.04005","version":2},"attestation_state":"computed","paper":{"title":"Qihoo-T2X: An Efficient Proxy-Tokenized Diffusion Transformer for Text-to-Any-Task","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ao Ma, Dawei Leng, Jiasong Feng, Jing Wang, Xiaodan Liang, Yuhui Yin","submitted_at":"2024-09-06T03:13:45Z","abstract_excerpt":"The global self-attention mechanism in diffusion transformers involves redundant computation due to the sparse and redundant nature of visual information, and the attention map of tokens within a spatial window shows significant similarity. To address this redundancy, we propose the Proxy-Tokenized Diffusion Transformer (PT-DiT), which employs sparse representative token attention (where the number of representative tokens is much smaller than the total number of tokens) to model global visual information efficiently. Specifically, within each transformer block, we compute an averaging token f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.04005","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-09-06T03:13:45Z","cross_cats_sorted":[],"title_canon_sha256":"574327368d2ef4cd36546a413bfe22de18522de1ac33d48cf2b6b59e51eeff57","abstract_canon_sha256":"2fe369ac4c4267266f2ed5f78600cbe1a718930fd15e3ba5d97d0b1fc4f68594"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:59.253219Z","signature_b64":"QKTmenD7Mkmgx4AOLBm7rgFpYln9WQSu7Z0/jaVm3536fZqa3wd82on4SDpqXrsMMwj+d/5xHEnsnfP6FqE1Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a04a3981383df9a22e609143579ef45e64613df0f4358de9bf04a47b3f93b77d","last_reissued_at":"2026-07-05T09:15:59.252701Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:59.252701Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Qihoo-T2X: An Efficient Proxy-Tokenized Diffusion Transformer for Text-to-Any-Task","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ao Ma, Dawei Leng, Jiasong Feng, Jing Wang, Xiaodan Liang, Yuhui Yin","submitted_at":"2024-09-06T03:13:45Z","abstract_excerpt":"The global self-attention mechanism in diffusion transformers involves redundant computation due to the sparse and redundant nature of visual information, and the attention map of tokens within a spatial window shows significant similarity. To address this redundancy, we propose the Proxy-Tokenized Diffusion Transformer (PT-DiT), which employs sparse representative token attention (where the number of representative tokens is much smaller than the total number of tokens) to model global visual information efficiently. Specifically, within each transformer block, we compute an averaging token f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.04005","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.04005/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.04005","created_at":"2026-07-05T09:15:59.252770+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.04005v2","created_at":"2026-07-05T09:15:59.252770+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.04005","created_at":"2026-07-05T09:15:59.252770+00:00"},{"alias_kind":"pith_short_12","alias_value":"UBFDTAJYHX42","created_at":"2026-07-05T09:15:59.252770+00:00"},{"alias_kind":"pith_short_16","alias_value":"UBFDTAJYHX42ELTA","created_at":"2026-07-05T09:15:59.252770+00:00"},{"alias_kind":"pith_short_8","alias_value":"UBFDTAJY","created_at":"2026-07-05T09:15:59.252770+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.06155","citing_title":"Efficient-vDiT: Efficient Video Diffusion Transformers With Attention Tile","ref_index":50,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UBFDTAJYHX42ELTASFBVPHXULZ","json":"https://pith.science/pith/UBFDTAJYHX42ELTASFBVPHXULZ.json","graph_json":"https://pith.science/api/pith-number/UBFDTAJYHX42ELTASFBVPHXULZ/graph.json","events_json":"https://pith.science/api/pith-number/UBFDTAJYHX42ELTASFBVPHXULZ/events.json","paper":"https://pith.science/paper/UBFDTAJY"},"agent_actions":{"view_html":"https://pith.science/pith/UBFDTAJYHX42ELTASFBVPHXULZ","download_json":"https://pith.science/pith/UBFDTAJYHX42ELTASFBVPHXULZ.json","view_paper":"https://pith.science/paper/UBFDTAJY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.04005&json=true","fetch_graph":"https://pith.science/api/pith-number/UBFDTAJYHX42ELTASFBVPHXULZ/graph.json","fetch_events":"https://pith.science/api/pith-number/UBFDTAJYHX42ELTASFBVPHXULZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UBFDTAJYHX42ELTASFBVPHXULZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UBFDTAJYHX42ELTASFBVPHXULZ/action/storage_attestation","attest_author":"https://pith.science/pith/UBFDTAJYHX42ELTASFBVPHXULZ/action/author_attestation","sign_citation":"https://pith.science/pith/UBFDTAJYHX42ELTASFBVPHXULZ/action/citation_signature","submit_replication":"https://pith.science/pith/UBFDTAJYHX42ELTASFBVPHXULZ/action/replication_record"}},"created_at":"2026-07-05T09:15:59.252770+00:00","updated_at":"2026-07-05T09:15:59.252770+00:00"}