{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CG76JTO7KIOWE7L4XI5MMOWC76","short_pith_number":"pith:CG76JTO7","schema_version":"1.0","canonical_sha256":"11bfe4cddf521d627d7cba3ac63ac2ffa9b4de95176707ad25c2c9dbcd13831f","source":{"kind":"arxiv","id":"2505.17017","version":2},"attestation_state":"computed","paper":{"title":"Delving into RL for Image Generation with CoT: A Study on DPO vs. GRPO","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chengzhuo Tong, Hongsheng Li, Pheng-Ann Heng, Renrui Zhang, Wenyu Shan, Xinyu Wei, Zhenghao Xing, Ziyu Guo","submitted_at":"2025-05-22T17:59:49Z","abstract_excerpt":"Recent advancements underscore the significant role of Reinforcement Learning (RL) in enhancing the Chain-of-Thought (CoT) reasoning capabilities of large language models (LLMs). Two prominent RL algorithms, Direct Preference Optimization (DPO) and Group Relative Policy Optimization (GRPO), are central to these developments, showcasing different pros and cons. Autoregressive image generation, also interpretable as a sequential CoT reasoning process, presents unique challenges distinct from LLM-based CoT reasoning. These encompass ensuring text-image consistency, improving image aesthetic quali"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.17017","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-22T17:59:49Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"62e56fa00ca118add57fbf09fa74effb61fad663d0963aab9fd8f552dd69db61","abstract_canon_sha256":"0f58350da6ad3065e35b14fe08fd91f4ee1aea10f409b5debe3cb8e387232275"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:07.430600Z","signature_b64":"hOCyJMg5HHVPa18KwUl7p51LCCI59m9l1GqoOGg/YKsorNSMhjf/Vss5kgQMCJh6quBjspqfFY/xyBIswsy+AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"11bfe4cddf521d627d7cba3ac63ac2ffa9b4de95176707ad25c2c9dbcd13831f","last_reissued_at":"2026-07-05T11:19:07.430106Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:07.430106Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Delving into RL for Image Generation with CoT: A Study on DPO vs. GRPO","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chengzhuo Tong, Hongsheng Li, Pheng-Ann Heng, Renrui Zhang, Wenyu Shan, Xinyu Wei, Zhenghao Xing, Ziyu Guo","submitted_at":"2025-05-22T17:59:49Z","abstract_excerpt":"Recent advancements underscore the significant role of Reinforcement Learning (RL) in enhancing the Chain-of-Thought (CoT) reasoning capabilities of large language models (LLMs). Two prominent RL algorithms, Direct Preference Optimization (DPO) and Group Relative Policy Optimization (GRPO), are central to these developments, showcasing different pros and cons. Autoregressive image generation, also interpretable as a sequential CoT reasoning process, presents unique challenges distinct from LLM-based CoT reasoning. These encompass ensuring text-image consistency, improving image aesthetic quali"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.17017","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.17017/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.17017","created_at":"2026-07-05T11:19:07.430163+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.17017v2","created_at":"2026-07-05T11:19:07.430163+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.17017","created_at":"2026-07-05T11:19:07.430163+00:00"},{"alias_kind":"pith_short_12","alias_value":"CG76JTO7KIOW","created_at":"2026-07-05T11:19:07.430163+00:00"},{"alias_kind":"pith_short_16","alias_value":"CG76JTO7KIOWE7L4","created_at":"2026-07-05T11:19:07.430163+00:00"},{"alias_kind":"pith_short_8","alias_value":"CG76JTO7","created_at":"2026-07-05T11:19:07.430163+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17888","citing_title":"MathVis-Fine: Aligning Visual Supervision with Necessity via Progressive Dependency-Guided Training for Multimodal Mathematical Reasoning","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02291","citing_title":"Optimizing Visual Generative Models via Distribution-wise Rewards","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2506.18871","citing_title":"OmniGen2: Towards Instruction-Aligned Multimodal Generation","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2508.20751","citing_title":"Pref-GRPO: Pairwise Preference Reward-based GRPO for Stable Text-to-Image Reinforcement Learning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2512.07348","citing_title":"MICo-150K: A Comprehensive Dataset Advancing Multi-Image Composition","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02355","citing_title":"From Broad Exploration to Stable Synthesis: Entropy-Guided Optimization for Autoregressive Image Generation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09629","citing_title":"HumorGen: Cognitive Synergy for Humor Generation in Large Language Models via Persona-Based Distillation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02467","citing_title":"VERTIGO: Visual Preference Optimization for Cinematic Camera Trajectory Generation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21802","citing_title":"MixGRPO: Unlocking Flow-based GRPO Efficiency with Mixed ODE-SDE","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CG76JTO7KIOWE7L4XI5MMOWC76","json":"https://pith.science/pith/CG76JTO7KIOWE7L4XI5MMOWC76.json","graph_json":"https://pith.science/api/pith-number/CG76JTO7KIOWE7L4XI5MMOWC76/graph.json","events_json":"https://pith.science/api/pith-number/CG76JTO7KIOWE7L4XI5MMOWC76/events.json","paper":"https://pith.science/paper/CG76JTO7"},"agent_actions":{"view_html":"https://pith.science/pith/CG76JTO7KIOWE7L4XI5MMOWC76","download_json":"https://pith.science/pith/CG76JTO7KIOWE7L4XI5MMOWC76.json","view_paper":"https://pith.science/paper/CG76JTO7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.17017&json=true","fetch_graph":"https://pith.science/api/pith-number/CG76JTO7KIOWE7L4XI5MMOWC76/graph.json","fetch_events":"https://pith.science/api/pith-number/CG76JTO7KIOWE7L4XI5MMOWC76/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CG76JTO7KIOWE7L4XI5MMOWC76/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CG76JTO7KIOWE7L4XI5MMOWC76/action/storage_attestation","attest_author":"https://pith.science/pith/CG76JTO7KIOWE7L4XI5MMOWC76/action/author_attestation","sign_citation":"https://pith.science/pith/CG76JTO7KIOWE7L4XI5MMOWC76/action/citation_signature","submit_replication":"https://pith.science/pith/CG76JTO7KIOWE7L4XI5MMOWC76/action/replication_record"}},"created_at":"2026-07-05T11:19:07.430163+00:00","updated_at":"2026-07-05T11:19:07.430163+00:00"}