{"work":{"id":"d06d7ecc-7579-4f89-a60b-4278a0f3c562","openalex_id":"https://openalex.org/W6966874861","doi":"10.48550/arxiv.2508.02324","arxiv_id":"2508.02324","raw_key":null,"title":"Qwen-Image Technical Report","authors":null,"authors_text":"Chenfei Wu, Jiahao Li, Jingren Zhou, Junyang Lin, Kaiyuan Gao, Kun Yan","year":2025,"venue":"cs.CV","abstract":"We present Qwen-Image, an image generation foundation model in the Qwen series that achieves significant advances in complex text rendering and precise image editing. To address the challenges of complex text rendering, we design a comprehensive data pipeline that includes large-scale data collection, filtering, annotation, synthesis, and balancing. Moreover, we adopt a progressive training strategy that starts with non-text-to-text rendering, evolves from simple to complex textual inputs, and gradually scales up to paragraph-level descriptions. This curriculum learning approach substantially enhances the model's native text rendering capabilities. As a result, Qwen-Image not only performs exceptionally well in alphabetic languages such as English, but also achieves remarkable progress on more challenging logographic languages like Chinese. To enhance image editing consistency, we introduce an improved multi-task training paradigm that incorporates not only traditional text-to-image (T2I) and text-image-to-image (TI2I) tasks but also image-to-image (I2I) reconstruction, effectively aligning the latent representations between Qwen2.5-VL and MMDiT. Furthermore, we separately feed the original image into Qwen2.5-VL and the VAE encoder to obtain semantic and reconstructive representations, respectively. This dual-encoding mechanism enables the editing module to strike a balance between preserving semantic consistency and maintaining visual fidelity. Qwen-Image achieves state-of-the-art performance, demonstrating its strong capabilities in both image generation and editing across multiple benchmarks.","external_url":"https://arxiv.org/abs/2508.02324","cited_by_count":0,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"2508.02324","created_at":"2026-05-08T21:39:25.073396+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":false,"display_title":"Qwen-Image Technical Report","render_title":"Qwen-Image Technical Report"},"hub":{"state":{"work_id":"d06d7ecc-7579-4f89-a60b-4278a0f3c562","tier":"super_hub","tier_reason":"100+ Pith inbound or 10,000+ external citations","pith_inbound_count":354,"external_cited_by_count":0,"distinct_field_count":10,"first_pith_cited_at":"2025-03-10T12:47:53+00:00","last_pith_cited_at":"2026-07-09T07:59:29+00:00","author_build_status":"needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-21T10:29:22.557801+00:00","tier_text":"super_hub"},"tier":"super_hub","role_counts":[{"context_role":"background","n":34},{"context_role":"baseline","n":23},{"context_role":"method","n":10},{"context_role":"dataset","n":2}],"polarity_counts":[{"context_polarity":"background","n":32},{"context_polarity":"baseline","n":23},{"context_polarity":"use_method","n":10},{"context_polarity":"unclear","n":2},{"context_polarity":"use_dataset","n":2}],"runs":{"ask_index":{"job_type":"ask_index","status":"succeeded","result":{"title":"Qwen-Image Technical Report","claims":[{"claim_text":"We present Qwen-Image, an image generation foundation model in the Qwen series that achieves significant advances in complex text rendering and precise image editing. To address the challenges of complex text rendering, we design a comprehensive data pipeline that includes large-scale data collection, filtering, annotation, synthesis, and balancing. Moreover, we adopt a progressive training strategy that starts with non-text-to-text rendering, evolves from simple to complex textual inputs, and gradually scales up to paragraph-level descriptions. This curriculum learning approach substantially ","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Qwen-Image Technical Report because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-14T00:44:22.303017+00:00"},"author_expand":{"job_type":"author_expand","status":"succeeded","result":{"authors_linked":[{"id":"4149062b-620e-4841-be0f-5dba072ecd09","orcid":null,"display_name":"Chenfei Wu"},{"id":"70bc88ce-9414-4600-8a49-588c93533fa5","orcid":null,"display_name":"Jiahao Li"},{"id":"12b62272-ac76-4164-95c5-eea2b030323c","orcid":null,"display_name":"Jingren Zhou"},{"id":"a59a1eb1-c0ef-4ae2-a0a7-765219185a00","orcid":null,"display_name":"Junyang Lin"},{"id":"360f3ff0-2318-4dc0-b6cf-8b70bc9effa3","orcid":null,"display_name":"Kaiyuan Gao"},{"id":"ae10eb6a-0269-4afc-bc9c-6f120fa7d4d5","orcid":null,"display_name":"Kun Yan"}]},"error":null,"updated_at":"2026-05-14T00:44:17.002923+00:00"},"context_extract":{"job_type":"context_extract","status":"succeeded","result":{"enqueued_papers":25},"error":null,"updated_at":"2026-05-14T00:44:22.179695+00:00"},"graph_features":{"job_type":"graph_features","status":"succeeded","result":{"co_cited":[{"title":"FLUX.1 Kontext: Flow Matching for In-Context Image Generation and Editing in Latent Space","work_id":"5dfe19d5-3541-4803-8fe9-3c8b9e29b281","shared_citers":46},{"title":"Emerging Properties in Unified Multimodal Pretraining","work_id":"e0cfd82c-f5d4-44fd-b531-ec73ab0a805b","shared_citers":33},{"title":"Step1X-Edit: A Practical Framework for General Image Editing","work_id":"3392f2c8-a1cb-4d6c-8c82-2cdccffa33f9","shared_citers":29},{"title":"OmniGen2: Towards Instruction-Aligned Multimodal Generation","work_id":"d3153e5f-b6e2-4ab3-9f41-e24e24d64496","shared_citers":28},{"title":"Qwen3-VL Technical Report","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","shared_citers":27},{"title":"Wan: Open and Advanced Large-Scale Video Generative Models","work_id":"ad3ebc3b-4224-46c9-b61d-bcf135da0a7c","shared_citers":27},{"title":"SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis","work_id":"8034c587-fba6-4941-87ba-c98f2ac962cb","shared_citers":26},{"title":"Qwen2.5-VL Technical Report","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","shared_citers":25},{"title":"ImgEdit: A Unified Image Editing Dataset and Benchmark","work_id":"059b5c3a-404c-4d30-a631-68c1d88a08a7","shared_citers":20},{"title":"Seedream 4.0: Toward Next-generation Multimodal Image Generation","work_id":"15c839a0-48a3-4218-82b6-cac5b7f66e13","shared_citers":20},{"title":"Z-Image: An Efficient Image Generation Foundation Model with Single-Stream Diffusion Transformer","work_id":"f1080a62-48e1-4255-b023-7556be57370d","shared_citers":19},{"title":"ELLA: Equip Diffusion Models with LLM for Enhanced Semantic Alignment","work_id":"94248955-4bc5-4517-98a0-66224a36d865","shared_citers":18},{"title":"Flow Matching for Generative Modeling","work_id":"6edb71c4-5d64-40af-a394-9757ea051a36","shared_citers":17},{"title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","work_id":"67d9e391-26d1-459e-ab56-07e60511c886","shared_citers":17},{"title":"UniWorld-V1: High-Resolution Semantic Encoders for Unified Visual Understanding and Generation","work_id":"488a273e-95d8-46f1-87c7-2244068d00d0","shared_citers":17},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":16},{"title":"BLIP3-o: A Family of Fully Open Unified Multimodal Models-Architecture, Training and Dataset","work_id":"86d896d2-592f-4d9b-938e-dfeb11f9388f","shared_citers":15},{"title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","shared_citers":15},{"title":"HunyuanVideo: A Systematic Framework For Large Video Generative Models","work_id":"881efa7e-7e73-4c66-9cc3-2803e551061c","shared_citers":15},{"title":"IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models","work_id":"98e51b10-54bd-4251-8a2d-f79bd6215c19","shared_citers":15},{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","shared_citers":15},{"title":"Denoising Diffusion Implicit Models","work_id":"8fa2128b-d18c-405c-ac92-0e669cf89ac0","shared_citers":14},{"title":"Hierarchical Text-Conditional Image Generation with CLIP Latents","work_id":"0c6a768b-70b8-4242-bb0e-459f1008c9fc","shared_citers":14},{"title":"Seedream 3.0 Technical Report","work_id":"013e56d0-7f47-4d0e-bbca-e9540fc0e0cc","shared_citers":14}],"time_series":[{"n":2,"year":2025},{"n":119,"year":2026}]},"error":null,"updated_at":"2026-05-14T00:44:16.256934+00:00"},"identity_refresh":{"job_type":"identity_refresh","status":"succeeded","result":{"fixed":1,"items":[{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","resolver":"local_arxiv","confidence":0.98,"old_work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e"}],"errors":[],"attempted":1},"error":null,"updated_at":"2026-05-14T00:44:26.939231+00:00"},"role_polarity":{"job_type":"role_polarity","status":"succeeded","result":{"title":"Qwen-Image Technical Report","claims":[{"claim_text":"We present Qwen-Image, an image generation foundation model in the Qwen series that achieves significant advances in complex text rendering and precise image editing. To address the challenges of complex text rendering, we design a comprehensive data pipeline that includes large-scale data collection, filtering, annotation, synthesis, and balancing. Moreover, we adopt a progressive training strategy that starts with non-text-to-text rendering, evolves from simple to complex textual inputs, and gradually scales up to paragraph-level descriptions. This curriculum learning approach substantially ","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Qwen-Image Technical Report because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-14T00:44:10.956561+00:00"},"summary_claims":{"job_type":"summary_claims","status":"succeeded","result":{"title":"Qwen-Image Technical Report","claims":[{"claim_text":"We present Qwen-Image, an image generation foundation model in the Qwen series that achieves significant advances in complex text rendering and precise image editing. To address the challenges of complex text rendering, we design a comprehensive data pipeline that includes large-scale data collection, filtering, annotation, synthesis, and balancing. Moreover, we adopt a progressive training strategy that starts with non-text-to-text rendering, evolves from simple to complex textual inputs, and gradually scales up to paragraph-level descriptions. This curriculum learning approach substantially ","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Qwen-Image Technical Report because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-14T00:44:15.194595+00:00"}},"summary":{"title":"Qwen-Image Technical Report","claims":[{"claim_text":"We present Qwen-Image, an image generation foundation model in the Qwen series that achieves significant advances in complex text rendering and precise image editing. To address the challenges of complex text rendering, we design a comprehensive data pipeline that includes large-scale data collection, filtering, annotation, synthesis, and balancing. Moreover, we adopt a progressive training strategy that starts with non-text-to-text rendering, evolves from simple to complex textual inputs, and gradually scales up to paragraph-level descriptions. This curriculum learning approach substantially ","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Qwen-Image Technical Report because it crossed a citation-hub threshold.","role_counts":[]},"graph":{"co_cited":[{"title":"FLUX.1 Kontext: Flow Matching for In-Context Image Generation and Editing in Latent Space","work_id":"5dfe19d5-3541-4803-8fe9-3c8b9e29b281","shared_citers":46},{"title":"Emerging Properties in Unified Multimodal Pretraining","work_id":"e0cfd82c-f5d4-44fd-b531-ec73ab0a805b","shared_citers":33},{"title":"Step1X-Edit: A Practical Framework for General Image Editing","work_id":"3392f2c8-a1cb-4d6c-8c82-2cdccffa33f9","shared_citers":29},{"title":"OmniGen2: Towards Instruction-Aligned Multimodal Generation","work_id":"d3153e5f-b6e2-4ab3-9f41-e24e24d64496","shared_citers":28},{"title":"Qwen3-VL Technical Report","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","shared_citers":27},{"title":"Wan: Open and Advanced Large-Scale Video Generative Models","work_id":"ad3ebc3b-4224-46c9-b61d-bcf135da0a7c","shared_citers":27},{"title":"SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis","work_id":"8034c587-fba6-4941-87ba-c98f2ac962cb","shared_citers":26},{"title":"Qwen2.5-VL Technical Report","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","shared_citers":25},{"title":"ImgEdit: A Unified Image Editing Dataset and Benchmark","work_id":"059b5c3a-404c-4d30-a631-68c1d88a08a7","shared_citers":20},{"title":"Seedream 4.0: Toward Next-generation Multimodal Image Generation","work_id":"15c839a0-48a3-4218-82b6-cac5b7f66e13","shared_citers":20},{"title":"Z-Image: An Efficient Image Generation Foundation Model with Single-Stream Diffusion Transformer","work_id":"f1080a62-48e1-4255-b023-7556be57370d","shared_citers":19},{"title":"ELLA: Equip Diffusion Models with LLM for Enhanced Semantic Alignment","work_id":"94248955-4bc5-4517-98a0-66224a36d865","shared_citers":18},{"title":"Flow Matching for Generative Modeling","work_id":"6edb71c4-5d64-40af-a394-9757ea051a36","shared_citers":17},{"title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","work_id":"67d9e391-26d1-459e-ab56-07e60511c886","shared_citers":17},{"title":"UniWorld-V1: High-Resolution Semantic Encoders for Unified Visual Understanding and Generation","work_id":"488a273e-95d8-46f1-87c7-2244068d00d0","shared_citers":17},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":16},{"title":"BLIP3-o: A Family of Fully Open Unified Multimodal Models-Architecture, Training and Dataset","work_id":"86d896d2-592f-4d9b-938e-dfeb11f9388f","shared_citers":15},{"title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","shared_citers":15},{"title":"HunyuanVideo: A Systematic Framework For Large Video Generative Models","work_id":"881efa7e-7e73-4c66-9cc3-2803e551061c","shared_citers":15},{"title":"IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models","work_id":"98e51b10-54bd-4251-8a2d-f79bd6215c19","shared_citers":15},{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","shared_citers":15},{"title":"Denoising Diffusion Implicit Models","work_id":"8fa2128b-d18c-405c-ac92-0e669cf89ac0","shared_citers":14},{"title":"Hierarchical Text-Conditional Image Generation with CLIP Latents","work_id":"0c6a768b-70b8-4242-bb0e-459f1008c9fc","shared_citers":14},{"title":"Seedream 3.0 Technical Report","work_id":"013e56d0-7f47-4d0e-bbca-e9540fc0e0cc","shared_citers":14}],"time_series":[{"n":2,"year":2025},{"n":119,"year":2026}]},"authors":[{"id":"4149062b-620e-4841-be0f-5dba072ecd09","orcid":null,"display_name":"Chenfei Wu","source":"manual","import_confidence":0.72},{"id":"70bc88ce-9414-4600-8a49-588c93533fa5","orcid":null,"display_name":"Jiahao Li","source":"manual","import_confidence":0.72},{"id":"12b62272-ac76-4164-95c5-eea2b030323c","orcid":null,"display_name":"Jingren Zhou","source":"manual","import_confidence":0.72},{"id":"a59a1eb1-c0ef-4ae2-a0a7-765219185a00","orcid":null,"display_name":"Junyang Lin","source":"manual","import_confidence":0.72},{"id":"360f3ff0-2318-4dc0-b6cf-8b70bc9effa3","orcid":null,"display_name":"Kaiyuan Gao","source":"manual","import_confidence":0.72},{"id":"ae10eb6a-0269-4afc-bc9c-6f120fa7d4d5","orcid":null,"display_name":"Kun Yan","source":"manual","import_confidence":0.72}]}}