{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:CWCQ2R6H5IEQSO6GJQZ55DAMD3","short_pith_number":"pith:CWCQ2R6H","schema_version":"1.0","canonical_sha256":"15850d47c7ea09093bc64c33de8c0c1edd39f092395d7730051fec11be3b334a","source":{"kind":"arxiv","id":"2012.09841","version":3},"attestation_state":"computed","paper":{"title":"Taming Transformers for High-Resolution Image Synthesis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bj\\\"orn Ommer, Patrick Esser, Robin Rombach","submitted_at":"2020-12-17T18:57:28Z","abstract_excerpt":"Designed to learn long-range interactions on sequential data, transformers continue to show state-of-the-art results on a wide variety of tasks. In contrast to CNNs, they contain no inductive bias that prioritizes local interactions. This makes them expressive, but also computationally infeasible for long sequences, such as high-resolution images. We demonstrate how combining the effectiveness of the inductive bias of CNNs with the expressivity of transformers enables them to model and thereby synthesize high-resolution images. We show how to (i) use CNNs to learn a context-rich vocabulary of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2012.09841","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2020-12-17T18:57:28Z","cross_cats_sorted":[],"title_canon_sha256":"670209fb29ed4bd6109f5fc2b87d30c6dec9cfe59ea4ec73b6ff32b0c6f13a94","abstract_canon_sha256":"c16e1eacdd3c0b493ffc3d3b75d19b3cf58d99bae51b9d2d343a70072b5c0494"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:51:53.958609Z","signature_b64":"nXW15OwaYthX+aCQyNWaDy3sfRLgpK82n8twVS0gimXt8f3zw2b7HPUGdsEFzni1OUVGJQJuF3z644pDqSvbAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"15850d47c7ea09093bc64c33de8c0c1edd39f092395d7730051fec11be3b334a","last_reissued_at":"2026-07-05T02:51:53.958145Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:51:53.958145Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Taming Transformers for High-Resolution Image Synthesis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bj\\\"orn Ommer, Patrick Esser, Robin Rombach","submitted_at":"2020-12-17T18:57:28Z","abstract_excerpt":"Designed to learn long-range interactions on sequential data, transformers continue to show state-of-the-art results on a wide variety of tasks. In contrast to CNNs, they contain no inductive bias that prioritizes local interactions. This makes them expressive, but also computationally infeasible for long sequences, such as high-resolution images. We demonstrate how combining the effectiveness of the inductive bias of CNNs with the expressivity of transformers enables them to model and thereby synthesize high-resolution images. We show how to (i) use CNNs to learn a context-rich vocabulary of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.09841","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.09841/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2012.09841","created_at":"2026-07-05T02:51:53.958206+00:00"},{"alias_kind":"arxiv_version","alias_value":"2012.09841v3","created_at":"2026-07-05T02:51:53.958206+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.09841","created_at":"2026-07-05T02:51:53.958206+00:00"},{"alias_kind":"pith_short_12","alias_value":"CWCQ2R6H5IEQ","created_at":"2026-07-05T02:51:53.958206+00:00"},{"alias_kind":"pith_short_16","alias_value":"CWCQ2R6H5IEQSO6G","created_at":"2026-07-05T02:51:53.958206+00:00"},{"alias_kind":"pith_short_8","alias_value":"CWCQ2R6H","created_at":"2026-07-05T02:51:53.958206+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19151","citing_title":"The Market in the Model: Latent Diffusion as Neural Economy","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19195","citing_title":"Moebius: 0.2B Lightweight Image Inpainting Framework with 10B-Level Performance","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11363","citing_title":"NSVQ: Mitigating Codebook Collapse by Stabilizing Encoder Drift in Vector Quantization","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08032","citing_title":"Variational Proximal Policy Optimization","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25012","citing_title":"Learning from Semantic Dictionaries: Discriminative Codebook Contrastive Learning for Unified Visual Representation and Generation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30341","citing_title":"GPIC: A Giant Permissive Image Corpus for Visual Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02211","citing_title":"Consistency Training while Mitigating Obfuscation via Rate Matching","ref_index":141,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02305","citing_title":"Mapping Whisper Representations to Human ECoG Responses with Interpretable Time-Resolved Neural Encoding","ref_index":128,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12011","citing_title":"CaloArt: Large-Patch x-Prediction Diffusion Transformers for High-Granularity Calorimeter Shower Generation","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2112.10752","citing_title":"High-Resolution Image Synthesis with Latent Diffusion Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2205.15868","citing_title":"CogVideo: Large-scale Pretraining for Text-to-Video Generation via Transformers","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18869","citing_title":"Emu3: Next-Token Prediction is All You Need","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07915","citing_title":"What Matters for Diffusion-Friendly Latent Manifold? Prior-Aligned Autoencoders for Latent Diffusion","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2311.15127","citing_title":"Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2204.06125","citing_title":"Hierarchical Text-Conditional Image Generation with CLIP Latents","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CWCQ2R6H5IEQSO6GJQZ55DAMD3","json":"https://pith.science/pith/CWCQ2R6H5IEQSO6GJQZ55DAMD3.json","graph_json":"https://pith.science/api/pith-number/CWCQ2R6H5IEQSO6GJQZ55DAMD3/graph.json","events_json":"https://pith.science/api/pith-number/CWCQ2R6H5IEQSO6GJQZ55DAMD3/events.json","paper":"https://pith.science/paper/CWCQ2R6H"},"agent_actions":{"view_html":"https://pith.science/pith/CWCQ2R6H5IEQSO6GJQZ55DAMD3","download_json":"https://pith.science/pith/CWCQ2R6H5IEQSO6GJQZ55DAMD3.json","view_paper":"https://pith.science/paper/CWCQ2R6H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2012.09841&json=true","fetch_graph":"https://pith.science/api/pith-number/CWCQ2R6H5IEQSO6GJQZ55DAMD3/graph.json","fetch_events":"https://pith.science/api/pith-number/CWCQ2R6H5IEQSO6GJQZ55DAMD3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CWCQ2R6H5IEQSO6GJQZ55DAMD3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CWCQ2R6H5IEQSO6GJQZ55DAMD3/action/storage_attestation","attest_author":"https://pith.science/pith/CWCQ2R6H5IEQSO6GJQZ55DAMD3/action/author_attestation","sign_citation":"https://pith.science/pith/CWCQ2R6H5IEQSO6GJQZ55DAMD3/action/citation_signature","submit_replication":"https://pith.science/pith/CWCQ2R6H5IEQSO6GJQZ55DAMD3/action/replication_record"}},"created_at":"2026-07-05T02:51:53.958206+00:00","updated_at":"2026-07-05T02:51:53.958206+00:00"}