{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7MOJFQSGUZS5INDMRIMD43S4PZ","short_pith_number":"pith:7MOJFQSG","schema_version":"1.0","canonical_sha256":"fb1c92c246a665d4346c8a183e6e5c7e7f0a8d0998d49fe359c42a8e658258af","source":{"kind":"arxiv","id":"2508.06924","version":1},"attestation_state":"computed","paper":{"title":"AR-GRPO: Training Autoregressive Image Generation Models via Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fuzheng Zhang, Guorui Zhou, Jingyuan Zhang, Qi Wang, Shihao Yuan, Wangmeng Zuo, Yahui Liu, Yang Yue","submitted_at":"2025-08-09T10:37:26Z","abstract_excerpt":"Inspired by the success of reinforcement learning (RL) in refining large language models (LLMs), we propose AR-GRPO, an approach to integrate online RL training into autoregressive (AR) image generation models. We adapt the Group Relative Policy Optimization (GRPO) algorithm to refine the vanilla autoregressive models' outputs by carefully designed reward functions that evaluate generated images across multiple quality dimensions, including perceptual quality, realism, and semantic fidelity. We conduct comprehensive experiments on both class-conditional (i.e., class-to-image) and text-conditio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.06924","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-08-09T10:37:26Z","cross_cats_sorted":[],"title_canon_sha256":"3b94ae56d4d125c07275acb006ed073e930d437d8098530ccef0361860b56d08","abstract_canon_sha256":"395d1d64461ca27651f3eb39d2397ae281d19e741d0bb7822834e5a3d1c1caf8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:51:37.702222Z","signature_b64":"ZiUdJZBC9lwnGTibQBCRblJY1Re86l5RS/LECyGA8jrZ2eH+1dfDUKL9Zk6BQT08GHin27/Z52Nt1WrTMoIIAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb1c92c246a665d4346c8a183e6e5c7e7f0a8d0998d49fe359c42a8e658258af","last_reissued_at":"2026-07-05T11:51:37.701734Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:51:37.701734Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AR-GRPO: Training Autoregressive Image Generation Models via Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fuzheng Zhang, Guorui Zhou, Jingyuan Zhang, Qi Wang, Shihao Yuan, Wangmeng Zuo, Yahui Liu, Yang Yue","submitted_at":"2025-08-09T10:37:26Z","abstract_excerpt":"Inspired by the success of reinforcement learning (RL) in refining large language models (LLMs), we propose AR-GRPO, an approach to integrate online RL training into autoregressive (AR) image generation models. We adapt the Group Relative Policy Optimization (GRPO) algorithm to refine the vanilla autoregressive models' outputs by carefully designed reward functions that evaluate generated images across multiple quality dimensions, including perceptual quality, realism, and semantic fidelity. We conduct comprehensive experiments on both class-conditional (i.e., class-to-image) and text-conditio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.06924","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.06924/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.06924","created_at":"2026-07-05T11:51:37.701792+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.06924v1","created_at":"2026-07-05T11:51:37.701792+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.06924","created_at":"2026-07-05T11:51:37.701792+00:00"},{"alias_kind":"pith_short_12","alias_value":"7MOJFQSGUZS5","created_at":"2026-07-05T11:51:37.701792+00:00"},{"alias_kind":"pith_short_16","alias_value":"7MOJFQSGUZS5INDM","created_at":"2026-07-05T11:51:37.701792+00:00"},{"alias_kind":"pith_short_8","alias_value":"7MOJFQSG","created_at":"2026-07-05T11:51:37.701792+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.08063","citing_title":"Flow-OPD: On-Policy Distillation for Flow Matching Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08063","citing_title":"Flow-OPD: On-Policy Distillation for Flow Matching Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16842","citing_title":"Sketch Then Paint: Hierarchical Reinforcement Learning for Diffusion Multi-Modal Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08063","citing_title":"Flow-OPD: On-Policy Distillation for Flow Matching Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08063","citing_title":"Flow-OPD: On-Policy Distillation for Flow Matching Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10937","citing_title":"Power Reinforcement Post-Training of Text-to-Image Models with Super-Linear Advantage Shaping","ref_index":120,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06966","citing_title":"MAR-GRPO: Stabilized GRPO for AR-diffusion Hybrid Image Generation","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08063","citing_title":"Flow-OPD: On-Policy Distillation for Flow Matching Models","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7MOJFQSGUZS5INDMRIMD43S4PZ","json":"https://pith.science/pith/7MOJFQSGUZS5INDMRIMD43S4PZ.json","graph_json":"https://pith.science/api/pith-number/7MOJFQSGUZS5INDMRIMD43S4PZ/graph.json","events_json":"https://pith.science/api/pith-number/7MOJFQSGUZS5INDMRIMD43S4PZ/events.json","paper":"https://pith.science/paper/7MOJFQSG"},"agent_actions":{"view_html":"https://pith.science/pith/7MOJFQSGUZS5INDMRIMD43S4PZ","download_json":"https://pith.science/pith/7MOJFQSGUZS5INDMRIMD43S4PZ.json","view_paper":"https://pith.science/paper/7MOJFQSG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.06924&json=true","fetch_graph":"https://pith.science/api/pith-number/7MOJFQSGUZS5INDMRIMD43S4PZ/graph.json","fetch_events":"https://pith.science/api/pith-number/7MOJFQSGUZS5INDMRIMD43S4PZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7MOJFQSGUZS5INDMRIMD43S4PZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7MOJFQSGUZS5INDMRIMD43S4PZ/action/storage_attestation","attest_author":"https://pith.science/pith/7MOJFQSGUZS5INDMRIMD43S4PZ/action/author_attestation","sign_citation":"https://pith.science/pith/7MOJFQSGUZS5INDMRIMD43S4PZ/action/citation_signature","submit_replication":"https://pith.science/pith/7MOJFQSGUZS5INDMRIMD43S4PZ/action/replication_record"}},"created_at":"2026-07-05T11:51:37.701792+00:00","updated_at":"2026-07-05T11:51:37.701792+00:00"}