{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BAXQJPSARPO7OCD25EXW3P5VYX","short_pith_number":"pith:BAXQJPSA","schema_version":"1.0","canonical_sha256":"082f04be408bddf7087ae92f6dbfb5c5eee13f57039f02a5b4f439a0cb8b13a6","source":{"kind":"arxiv","id":"2501.13926","version":2},"attestation_state":"computed","paper":{"title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Chengzhuo Tong, Haoquan Zhang, Hongsheng Li, Jiaming Liu, Manyuan Zhang, Peng Gao, Pheng-Ann Heng, Renrui Zhang, Rui Huang, Shanghang Zhang, Zhizheng Zhao, Ziyu Guo","submitted_at":"2025-01-23T18:59:43Z","abstract_excerpt":"Chain-of-Thought (CoT) reasoning has been extensively explored in large models to tackle complex understanding tasks. However, it still remains an open question whether such strategies can be applied to verifying and reinforcing image generation scenarios. In this paper, we provide the first comprehensive investigation of the potential of CoT reasoning to enhance autoregressive image generation. We focus on three techniques: scaling test-time computation for verification, aligning model preferences with Direct Preference Optimization (DPO), and integrating these techniques for complementary ef"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.13926","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-01-23T18:59:43Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"7eab649e13e47d113e5fca0d05189f3f71eb8ac7b2ae85b927a70506bfbc080a","abstract_canon_sha256":"01bee03ca222c12842ac26f4ac168ce71e536f9876ae5ced92c60583b05f8045"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:41:38.442441Z","signature_b64":"6UTItckAxAG68aR2NuTMgsnJtDbkjTLKZzzZJKOM9I2SMxr7DzqDCWrblFUPotNvKZ+5I/2VuExSQ0vCxIddBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"082f04be408bddf7087ae92f6dbfb5c5eee13f57039f02a5b4f439a0cb8b13a6","last_reissued_at":"2026-07-05T11:41:38.441957Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:41:38.441957Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can We Generate Images with CoT? Let's Verify and Reinforce Image Generation Step by Step","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Chengzhuo Tong, Haoquan Zhang, Hongsheng Li, Jiaming Liu, Manyuan Zhang, Peng Gao, Pheng-Ann Heng, Renrui Zhang, Rui Huang, Shanghang Zhang, Zhizheng Zhao, Ziyu Guo","submitted_at":"2025-01-23T18:59:43Z","abstract_excerpt":"Chain-of-Thought (CoT) reasoning has been extensively explored in large models to tackle complex understanding tasks. However, it still remains an open question whether such strategies can be applied to verifying and reinforcing image generation scenarios. In this paper, we provide the first comprehensive investigation of the potential of CoT reasoning to enhance autoregressive image generation. We focus on three techniques: scaling test-time computation for verification, aligning model preferences with Direct Preference Optimization (DPO), and integrating these techniques for complementary ef"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.13926","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.13926/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.13926","created_at":"2026-07-05T11:41:38.442011+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.13926v2","created_at":"2026-07-05T11:41:38.442011+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.13926","created_at":"2026-07-05T11:41:38.442011+00:00"},{"alias_kind":"pith_short_12","alias_value":"BAXQJPSARPO7","created_at":"2026-07-05T11:41:38.442011+00:00"},{"alias_kind":"pith_short_16","alias_value":"BAXQJPSARPO7OCD2","created_at":"2026-07-05T11:41:38.442011+00:00"},{"alias_kind":"pith_short_8","alias_value":"BAXQJPSA","created_at":"2026-07-05T11:41:38.442011+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24849","citing_title":"IV-CoT: Implicit Visual Chain-of-Thought for Structure-Aware Text-to-Image Generation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17888","citing_title":"MathVis-Fine: Aligning Visual Supervision with Necessity via Progressive Dependency-Guided Training for Multimodal Mathematical Reasoning","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08231","citing_title":"Test-Time Scaling in Multimodal Foundation Models: A Comprehensive Survey of Generation and Reasoning","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04264","citing_title":"UniCanvas: A Diffusion-base Unified Model for Text-in-Image Joint Generation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2506.03530","citing_title":"How Far Are We from Generating Missing Modalities with Foundation Models?","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2506.16796","citing_title":"RealSR-R1: Reinforcement Learning for Real-World Image Super-Resolution with Vision-Language Chain-of-Thought","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2506.18871","citing_title":"OmniGen2: Towards Instruction-Aligned Multimodal Generation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2512.07348","citing_title":"MICo-150K: A Comprehensive Dataset Advancing Multi-Image Composition","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02355","citing_title":"From Broad Exploration to Stable Synthesis: Entropy-Guided Optimization for Autoregressive Image Generation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":234,"is_internal_anchor":false},{"citing_arxiv_id":"2505.07818","citing_title":"DanceGRPO: Unleashing GRPO on Visual Generation","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BAXQJPSARPO7OCD25EXW3P5VYX","json":"https://pith.science/pith/BAXQJPSARPO7OCD25EXW3P5VYX.json","graph_json":"https://pith.science/api/pith-number/BAXQJPSARPO7OCD25EXW3P5VYX/graph.json","events_json":"https://pith.science/api/pith-number/BAXQJPSARPO7OCD25EXW3P5VYX/events.json","paper":"https://pith.science/paper/BAXQJPSA"},"agent_actions":{"view_html":"https://pith.science/pith/BAXQJPSARPO7OCD25EXW3P5VYX","download_json":"https://pith.science/pith/BAXQJPSARPO7OCD25EXW3P5VYX.json","view_paper":"https://pith.science/paper/BAXQJPSA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.13926&json=true","fetch_graph":"https://pith.science/api/pith-number/BAXQJPSARPO7OCD25EXW3P5VYX/graph.json","fetch_events":"https://pith.science/api/pith-number/BAXQJPSARPO7OCD25EXW3P5VYX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BAXQJPSARPO7OCD25EXW3P5VYX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BAXQJPSARPO7OCD25EXW3P5VYX/action/storage_attestation","attest_author":"https://pith.science/pith/BAXQJPSARPO7OCD25EXW3P5VYX/action/author_attestation","sign_citation":"https://pith.science/pith/BAXQJPSARPO7OCD25EXW3P5VYX/action/citation_signature","submit_replication":"https://pith.science/pith/BAXQJPSARPO7OCD25EXW3P5VYX/action/replication_record"}},"created_at":"2026-07-05T11:41:38.442011+00:00","updated_at":"2026-07-05T11:41:38.442011+00:00"}