{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:F4VFV6XZGZK7GKKTTQ2D3TZLSZ","short_pith_number":"pith:F4VFV6XZ","schema_version":"1.0","canonical_sha256":"2f2a5afaf93655f329539c343dcf2b9658c479d2c9d2ddaa627e7e0e9330ef46","source":{"kind":"arxiv","id":"2509.06945","version":2},"attestation_state":"computed","paper":{"title":"Interleaving Reasoning for Better Text-to-Image Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Junbo Qiao, Philip Torr, Qingyu Yin, Shaohui Lin, Shaosheng Cao, Shixiang Tang, Shuang Chen, Wanli Ouyang, Wenbo Hu, Wenxuan Huang, Xiaoman Wang, Yao Hu, Yu Cheng, Yue Guo, Yufan Shen, Yuntian Tang, Zhenfei Yin, Zheyong Xie","submitted_at":"2025-09-08T17:56:23Z","abstract_excerpt":"Unified multimodal understanding and generation models recently have achieve significant improvement in image generation capability, yet a large gap remains in instruction following and detail preservation compared to systems that tightly couple comprehension with generation such as GPT-4o. Motivated by recent advances in interleaving reasoning, we explore whether such reasoning can further improve Text-to-Image (T2I) generation. We introduce Interleaving Reasoning Generation (IRG), a framework that alternates between text-based thinking and image synthesis: the model first produces a text-bas"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.06945","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-09-08T17:56:23Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"3a3db5116b69540caf13573aed1cbd837b9e0ee53642a668b2ee42b9cd4787bf","abstract_canon_sha256":"c06762c0f5a301ccffc45bbf0f1526c6226a74204ab2863b873b4484b1c2f373"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:07:12.780610Z","signature_b64":"sctLLNsv2KolVIqwiGBV7mAf3TBGUmAGrk70xnr78q3R30vk8zCtpjM3JbjEQtrT8AJUjDt6MRxVYJbA+QRQCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2f2a5afaf93655f329539c343dcf2b9658c479d2c9d2ddaa627e7e0e9330ef46","last_reissued_at":"2026-07-05T12:07:12.780111Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:07:12.780111Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Interleaving Reasoning for Better Text-to-Image Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Junbo Qiao, Philip Torr, Qingyu Yin, Shaohui Lin, Shaosheng Cao, Shixiang Tang, Shuang Chen, Wanli Ouyang, Wenbo Hu, Wenxuan Huang, Xiaoman Wang, Yao Hu, Yu Cheng, Yue Guo, Yufan Shen, Yuntian Tang, Zhenfei Yin, Zheyong Xie","submitted_at":"2025-09-08T17:56:23Z","abstract_excerpt":"Unified multimodal understanding and generation models recently have achieve significant improvement in image generation capability, yet a large gap remains in instruction following and detail preservation compared to systems that tightly couple comprehension with generation such as GPT-4o. Motivated by recent advances in interleaving reasoning, we explore whether such reasoning can further improve Text-to-Image (T2I) generation. We introduce Interleaving Reasoning Generation (IRG), a framework that alternates between text-based thinking and image synthesis: the model first produces a text-bas"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.06945","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.06945/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.06945","created_at":"2026-07-05T12:07:12.780169+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.06945v2","created_at":"2026-07-05T12:07:12.780169+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.06945","created_at":"2026-07-05T12:07:12.780169+00:00"},{"alias_kind":"pith_short_12","alias_value":"F4VFV6XZGZK7","created_at":"2026-07-05T12:07:12.780169+00:00"},{"alias_kind":"pith_short_16","alias_value":"F4VFV6XZGZK7GKKT","created_at":"2026-07-05T12:07:12.780169+00:00"},{"alias_kind":"pith_short_8","alias_value":"F4VFV6XZ","created_at":"2026-07-05T12:07:12.780169+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24849","citing_title":"IV-CoT: Implicit Visual Chain-of-Thought for Structure-Aware Text-to-Image Generation","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11683","citing_title":"Reason, Then Re-reason: Cross-view Revisiting Improves Spatial Reasoning","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31504","citing_title":"SimpleSearch-VL: A Simple Recipe for Multimodal Agentic Deep Search","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08063","citing_title":"Flow-OPD: On-Policy Distillation for Flow Matching Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28805","citing_title":"OmniVerifier-M1: Multimodal Meta-Verifier with Explicit Structured Recalibration","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10784","citing_title":"TorchUMM: A Unified Multimodal Model Codebase for Evaluation, Analysis, and Post-training","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08063","citing_title":"Flow-OPD: On-Policy Distillation for Flow Matching Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17766","citing_title":"LatentUMM: Dual Latent Alignment for Unified Multimodal Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14876","citing_title":"Unlocking Complex Visual Generation via Closed-Loop Verified Reasoning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01070","citing_title":"How RL Unlocks the Aha Moment in Geometric Interleaved Reasoning","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08063","citing_title":"Flow-OPD: On-Policy Distillation for Flow Matching Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11400","citing_title":"UniPath: Adaptive Coordination of Understanding and Generation for Unified Multimodal Reasoning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08063","citing_title":"Flow-OPD: On-Policy Distillation for Flow Matching Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25636","citing_title":"Refinement via Regeneration: Enlarging Modification Space Boosts Image Refinement in Unified Multimodal Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24625","citing_title":"Meta-CoT: Enhancing Granularity and Generalization in Image Editing","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10784","citing_title":"TorchUMM: A Unified Multimodal Model Codebase for Evaluation, Analysis, and Post-training","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08063","citing_title":"Flow-OPD: On-Policy Distillation for Flow Matching Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08043","citing_title":"SCOPE: Structured Decomposition and Conditional Skill Orchestration for Complex Image Generation","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F4VFV6XZGZK7GKKTTQ2D3TZLSZ","json":"https://pith.science/pith/F4VFV6XZGZK7GKKTTQ2D3TZLSZ.json","graph_json":"https://pith.science/api/pith-number/F4VFV6XZGZK7GKKTTQ2D3TZLSZ/graph.json","events_json":"https://pith.science/api/pith-number/F4VFV6XZGZK7GKKTTQ2D3TZLSZ/events.json","paper":"https://pith.science/paper/F4VFV6XZ"},"agent_actions":{"view_html":"https://pith.science/pith/F4VFV6XZGZK7GKKTTQ2D3TZLSZ","download_json":"https://pith.science/pith/F4VFV6XZGZK7GKKTTQ2D3TZLSZ.json","view_paper":"https://pith.science/paper/F4VFV6XZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.06945&json=true","fetch_graph":"https://pith.science/api/pith-number/F4VFV6XZGZK7GKKTTQ2D3TZLSZ/graph.json","fetch_events":"https://pith.science/api/pith-number/F4VFV6XZGZK7GKKTTQ2D3TZLSZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F4VFV6XZGZK7GKKTTQ2D3TZLSZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F4VFV6XZGZK7GKKTTQ2D3TZLSZ/action/storage_attestation","attest_author":"https://pith.science/pith/F4VFV6XZGZK7GKKTTQ2D3TZLSZ/action/author_attestation","sign_citation":"https://pith.science/pith/F4VFV6XZGZK7GKKTTQ2D3TZLSZ/action/citation_signature","submit_replication":"https://pith.science/pith/F4VFV6XZGZK7GKKTTQ2D3TZLSZ/action/replication_record"}},"created_at":"2026-07-05T12:07:12.780169+00:00","updated_at":"2026-07-05T12:07:12.780169+00:00"}