{"work":{"id":"0a105815-ff2e-43ce-8566-966cdcae1af4","openalex_id":"https://openalex.org/W4283388932","doi":"10.48550/arxiv.2206.10789","arxiv_id":"2206.10789","raw_key":null,"title":"Scaling Autoregressive Models for Content-Rich Text-to-Image Generation","authors":null,"authors_text":"Jiahui Yu, Yuanzhong Xu, Jing Yu Koh, Thang Luong, Gunjan Baid, Zirui Wang","year":2022,"venue":"cs.CV","abstract":"We present the Pathways Autoregressive Text-to-Image (Parti) model, which generates high-fidelity photorealistic images and supports content-rich synthesis involving complex compositions and world knowledge. Parti treats text-to-image generation as a sequence-to-sequence modeling problem, akin to machine translation, with sequences of image tokens as the target outputs rather than text tokens in another language. This strategy can naturally tap into the rich body of prior work on large language models, which have seen continued advances in capabilities and performance through scaling data and model sizes. Our approach is simple: First, Parti uses a Transformer-based image tokenizer, ViT-VQGAN, to encode images as sequences of discrete tokens. Second, we achieve consistent quality improvements by scaling the encoder-decoder Transformer model up to 20B parameters, with a new state-of-the-art zero-shot FID score of 7.23 and finetuned FID score of 3.22 on MS-COCO. Our detailed analysis on Localized Narratives as well as PartiPrompts (P2), a new holistic benchmark of over 1600 English prompts, demonstrate the effectiveness of Parti across a wide variety of categories and difficulty aspects. We also explore and highlight limitations of our models in order to define and exemplify key areas of focus for further improvements. See https://parti.research.google/ for high-resolution images.","external_url":"https://arxiv.org/abs/2206.10789","cited_by_count":340,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"2206.10789","created_at":"2026-05-10T16:36:02.112700+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":true,"display_title":"Scaling Autoregressive Models for Content-Rich Text-to-Image Generation","render_title":"Scaling Autoregressive Models for Content-Rich Text-to-Image Generation"},"hub":{"state":{"work_id":"0a105815-ff2e-43ce-8566-966cdcae1af4","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":92,"external_cited_by_count":340,"distinct_field_count":11,"first_pith_cited_at":"2022-08-02T17:50:36+00:00","last_pith_cited_at":"2026-07-01T00:42:40+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-21T23:29:18.362907+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":14},{"context_role":"dataset","n":4}],"polarity_counts":[{"context_polarity":"background","n":11},{"context_polarity":"use_dataset","n":4},{"context_polarity":"unclear","n":2},{"context_polarity":"support","n":1}],"runs":{"context_extract":{"job_type":"context_extract","status":"succeeded","result":{"enqueued_papers":25},"error":null,"updated_at":"2026-05-14T17:59:59.589616+00:00"},"graph_features":{"job_type":"graph_features","status":"succeeded","result":{"co_cited":[{"title":"Hierarchical Text-Conditional Image Generation with CLIP Latents","work_id":"0c6a768b-70b8-4242-bb0e-459f1008c9fc","shared_citers":16},{"title":"Classifier-Free Diffusion Guidance","work_id":"acf2c588-c088-4a6c-938e-150ad7c666d7","shared_citers":12},{"title":"SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis","work_id":"8034c587-fba6-4941-87ba-c98f2ac962cb","shared_citers":11},{"title":"GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models","work_id":"34430d19-7919-48ce-88a5-17b3bfe2192e","shared_citers":10},{"title":"Denoising Diffusion Implicit Models","work_id":"8fa2128b-d18c-405c-ac92-0e669cf89ac0","shared_citers":9},{"title":"Auto-Encoding Variational Bayes","work_id":"97d95295-30e1-42b4-bbf6-85f0fa4edb44","shared_citers":8},{"title":"Human Preference Score v2: A Solid Benchmark for Evaluating Human Preferences of Text-to-Image Synthesis","work_id":"40702548-f094-4c67-a5db-a62f426f852e","shared_citers":7},{"title":"LLaMA: Open and Efficient Foundation Language Models","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","shared_citers":7},{"title":"Autoregressive Model Beats Diffusion: Llama for Scalable Image Generation","work_id":"41efe203-9377-4c63-b1d6-e499cd6e46f6","shared_citers":6},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":6},{"title":"Learning Transferable Visual Models From Natural Language Supervision","work_id":"6de86bb5-27bd-4d5c-8b89-967ebfc52659","shared_citers":6},{"title":"Photorealistic Text-to-Image Diffusion Models with Deep Language Understanding","work_id":"af16442b-a46f-469d-8818-c37b53a504c7","shared_citers":6},{"title":"Score-Based Generative Modeling through Stochastic Differential Equations","work_id":"d9110e53-a5d4-4794-a4c5-a575e91c31ad","shared_citers":6},{"title":"Decoupled Weight Decay Regularization","work_id":"07ef7360-d385-4033-83f7-8384a6325204","shared_citers":5},{"title":"Flow Matching for Generative Modeling","work_id":"6edb71c4-5d64-40af-a394-9757ea051a36","shared_citers":5},{"title":"Scaling Laws for Neural Language Models","work_id":"b7dd8749-9c45-4977-ab9b-64478dce1ae8","shared_citers":5},{"title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","shared_citers":4},{"title":"CLIPScore: A Reference-free Evaluation Metric for Image Captioning","work_id":"2dd7e0b7-c69c-4976-a406-12d4f5b18d14","shared_citers":4},{"title":"DanceGRPO: Unleashing GRPO on Visual Generation","work_id":"7404dd36-8f9c-478f-b089-ef9f8189c711","shared_citers":4},{"title":"ELLA: Equip Diffusion Models with LLM for Enhanced Semantic Alignment","work_id":"94248955-4bc5-4517-98a0-66224a36d865","shared_citers":4},{"title":"Gaussian Error Linear Units (GELUs)","work_id":"0466fd22-03a1-4a61-af0a-a900e77bb023","shared_citers":4},{"title":"IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models","work_id":"98e51b10-54bd-4251-8a2d-f79bd6215c19","shared_citers":4},{"title":"PixArt-$\\alpha$: Fast Training of Diffusion Transformer for Photorealistic Text-to-Image Synthesis","work_id":"77157568-e4be-4041-bb20-388177fc59d0","shared_citers":4},{"title":"Qwen Technical Report","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","shared_citers":4}],"time_series":[{"n":8,"year":2022},{"n":3,"year":2023},{"n":5,"year":2024},{"n":2,"year":2025},{"n":18,"year":2026}],"dependency_candidates":[]},"error":null,"updated_at":"2026-05-14T17:59:31.455061+00:00"},"identity_refresh":{"job_type":"identity_refresh","status":"succeeded","result":{"items":[{"title":"Qwen3 Technical Report","outcome":"unchanged","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","resolver":"local_arxiv","confidence":0.98,"old_work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e"}],"counts":{"fixed":0,"merged":0,"unchanged":1,"quarantined":0,"needs_external_resolution":0},"errors":[],"attempted":1},"error":null,"updated_at":"2026-05-14T18:00:37.089947+00:00"},"summary_claims":{"job_type":"summary_claims","status":"succeeded","result":{"title":"Scaling Autoregressive Models for Content-Rich Text-to-Image Generation","claims":[{"claim_text":"We present the Pathways Autoregressive Text-to-Image (Parti) model, which generates high-fidelity photorealistic images and supports content-rich synthesis involving complex compositions and world knowledge. Parti treats text-to-image generation as a sequence-to-sequence modeling problem, akin to machine translation, with sequences of image tokens as the target outputs rather than text tokens in another language. This strategy can naturally tap into the rich body of prior work on large language models, which have seen continued advances in capabilities and performance through scaling data and ","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Scaling Autoregressive Models for Content-Rich Text-to-Image Generation because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-14T17:59:31.387506+00:00"}},"summary":{"title":"Scaling Autoregressive Models for Content-Rich Text-to-Image Generation","claims":[{"claim_text":"We present the Pathways Autoregressive Text-to-Image (Parti) model, which generates high-fidelity photorealistic images and supports content-rich synthesis involving complex compositions and world knowledge. Parti treats text-to-image generation as a sequence-to-sequence modeling problem, akin to machine translation, with sequences of image tokens as the target outputs rather than text tokens in another language. This strategy can naturally tap into the rich body of prior work on large language models, which have seen continued advances in capabilities and performance through scaling data and ","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Scaling Autoregressive Models for Content-Rich Text-to-Image Generation because it crossed a citation-hub threshold.","role_counts":[]},"graph":{"co_cited":[{"title":"Hierarchical Text-Conditional Image Generation with CLIP Latents","work_id":"0c6a768b-70b8-4242-bb0e-459f1008c9fc","shared_citers":16},{"title":"Classifier-Free Diffusion Guidance","work_id":"acf2c588-c088-4a6c-938e-150ad7c666d7","shared_citers":12},{"title":"SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis","work_id":"8034c587-fba6-4941-87ba-c98f2ac962cb","shared_citers":11},{"title":"GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models","work_id":"34430d19-7919-48ce-88a5-17b3bfe2192e","shared_citers":10},{"title":"Denoising Diffusion Implicit Models","work_id":"8fa2128b-d18c-405c-ac92-0e669cf89ac0","shared_citers":9},{"title":"Auto-Encoding Variational Bayes","work_id":"97d95295-30e1-42b4-bbf6-85f0fa4edb44","shared_citers":8},{"title":"Human Preference Score v2: A Solid Benchmark for Evaluating Human Preferences of Text-to-Image Synthesis","work_id":"40702548-f094-4c67-a5db-a62f426f852e","shared_citers":7},{"title":"LLaMA: Open and Efficient Foundation Language Models","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","shared_citers":7},{"title":"Autoregressive Model Beats Diffusion: Llama for Scalable Image Generation","work_id":"41efe203-9377-4c63-b1d6-e499cd6e46f6","shared_citers":6},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":6},{"title":"Learning Transferable Visual Models From Natural Language Supervision","work_id":"6de86bb5-27bd-4d5c-8b89-967ebfc52659","shared_citers":6},{"title":"Photorealistic Text-to-Image Diffusion Models with Deep Language Understanding","work_id":"af16442b-a46f-469d-8818-c37b53a504c7","shared_citers":6},{"title":"Score-Based Generative Modeling through Stochastic Differential Equations","work_id":"d9110e53-a5d4-4794-a4c5-a575e91c31ad","shared_citers":6},{"title":"Decoupled Weight Decay Regularization","work_id":"07ef7360-d385-4033-83f7-8384a6325204","shared_citers":5},{"title":"Flow Matching for Generative Modeling","work_id":"6edb71c4-5d64-40af-a394-9757ea051a36","shared_citers":5},{"title":"Scaling Laws for Neural Language Models","work_id":"b7dd8749-9c45-4977-ab9b-64478dce1ae8","shared_citers":5},{"title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","shared_citers":4},{"title":"CLIPScore: A Reference-free Evaluation Metric for Image Captioning","work_id":"2dd7e0b7-c69c-4976-a406-12d4f5b18d14","shared_citers":4},{"title":"DanceGRPO: Unleashing GRPO on Visual Generation","work_id":"7404dd36-8f9c-478f-b089-ef9f8189c711","shared_citers":4},{"title":"ELLA: Equip Diffusion Models with LLM for Enhanced Semantic Alignment","work_id":"94248955-4bc5-4517-98a0-66224a36d865","shared_citers":4},{"title":"Gaussian Error Linear Units (GELUs)","work_id":"0466fd22-03a1-4a61-af0a-a900e77bb023","shared_citers":4},{"title":"IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models","work_id":"98e51b10-54bd-4251-8a2d-f79bd6215c19","shared_citers":4},{"title":"PixArt-$\\alpha$: Fast Training of Diffusion Transformer for Photorealistic Text-to-Image Synthesis","work_id":"77157568-e4be-4041-bb20-388177fc59d0","shared_citers":4},{"title":"Qwen Technical Report","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","shared_citers":4}],"time_series":[{"n":8,"year":2022},{"n":3,"year":2023},{"n":5,"year":2024},{"n":2,"year":2025},{"n":18,"year":2026}],"dependency_candidates":[]},"authors":[]}}