{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SPJUXULNN4JE4UQRFHHXZSCSAY","short_pith_number":"pith:SPJUXULN","schema_version":"1.0","canonical_sha256":"93d34bd16d6f124e521129cf7cc852060b2e2022d838b418a307d54ca561b405","source":{"kind":"arxiv","id":"2304.05977","version":4},"attestation_state":"computed","paper":{"title":"ImageReward: Learning and Evaluating Human Preferences for Text-to-Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Jiazheng Xu, Jie Tang, Ming Ding, Qinkai Li, Xiao Liu, Yuchen Wu, Yuxiao Dong, Yuxuan Tong","submitted_at":"2023-04-12T16:58:13Z","abstract_excerpt":"We present a comprehensive solution to learn and improve text-to-image models from human preference feedback. To begin with, we build ImageReward -- the first general-purpose text-to-image human preference reward model -- to effectively encode human preferences. Its training is based on our systematic annotation pipeline including rating and ranking, which collects 137k expert comparisons to date. In human evaluation, ImageReward outperforms existing scoring models and metrics, making it a promising automatic metric for evaluating text-to-image synthesis. On top of it, we propose Reward Feedba"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.05977","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-04-12T16:58:13Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"f2bea333ec06346ae3ca24704e7cd5a4403cecd391ec84abcf98b3cc6f950f2b","abstract_canon_sha256":"95921b1a4aaf28fe73ba5d3a75e3425e7a578af7d6e4b0d60a91b4f8190baae2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:28:27.842094Z","signature_b64":"xDpcLfaZ7Pl7DyPeEyPrYlJKD6ukbfJ70vTUYWs+puiHmvsLAQ/50dyffC0P0Uk+wWjuaQaxjnV2vAAbbTUsAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"93d34bd16d6f124e521129cf7cc852060b2e2022d838b418a307d54ca561b405","last_reissued_at":"2026-07-05T07:28:27.841669Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:28:27.841669Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ImageReward: Learning and Evaluating Human Preferences for Text-to-Image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Jiazheng Xu, Jie Tang, Ming Ding, Qinkai Li, Xiao Liu, Yuchen Wu, Yuxiao Dong, Yuxuan Tong","submitted_at":"2023-04-12T16:58:13Z","abstract_excerpt":"We present a comprehensive solution to learn and improve text-to-image models from human preference feedback. To begin with, we build ImageReward -- the first general-purpose text-to-image human preference reward model -- to effectively encode human preferences. Its training is based on our systematic annotation pipeline including rating and ranking, which collects 137k expert comparisons to date. In human evaluation, ImageReward outperforms existing scoring models and metrics, making it a promising automatic metric for evaluating text-to-image synthesis. On top of it, we propose Reward Feedba"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.05977","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.05977/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.05977","created_at":"2026-07-05T07:28:27.841723+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.05977v4","created_at":"2026-07-05T07:28:27.841723+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.05977","created_at":"2026-07-05T07:28:27.841723+00:00"},{"alias_kind":"pith_short_12","alias_value":"SPJUXULNN4JE","created_at":"2026-07-05T07:28:27.841723+00:00"},{"alias_kind":"pith_short_16","alias_value":"SPJUXULNN4JE4UQR","created_at":"2026-07-05T07:28:27.841723+00:00"},{"alias_kind":"pith_short_8","alias_value":"SPJUXULN","created_at":"2026-07-05T07:28:27.841723+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22394","citing_title":"Curvature-Adaptive Consistency Flow Matching: Autonomous Trajectory Optimization via Reinforcement Learning","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01803","citing_title":"PixGS: Pixel-Space Diffusion for Direct 3D Gaussian Splat Generation","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27147","citing_title":"How to Guide Your Flow: Few-Step Alignment via Flow Map Reward Guidance","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29059","citing_title":"Flow Matching in Feature Space for Stochastic World Modeling","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28615","citing_title":"Compositional Text-to-Image Generation Via Region-aware Bimodal Direct Preference Optimization","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00583","citing_title":"Improving Visual Representation Alignment Generation with GRPO","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27147","citing_title":"How to Guide Your Flow: Few-Step Alignment via Flow Map Reward Guidance","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16951","citing_title":"Edit-GRPO: A Locality-Preserving Policy Optimization Framework for Image Editing","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2506.22832","citing_title":"Listener-Rewarded Thinking in VLMs for Image Preferences","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06779","citing_title":"VASR: Variance-Aware Systematic Resampling for Reward-Guided Diffusion","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27147","citing_title":"How to Guide Your Flow: Few-Step Alignment via Flow Map Reward Guidance","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26503","citing_title":"Delta Score Matters! Spatial Adaptive Multi Guidance in Diffusion Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23536","citing_title":"$Z^2$-Sampling: Zero-Cost Zigzag Trajectories for Semantic Alignment in Diffusion Models","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23540","citing_title":"Oracle Noise: Faster Semantic Spherical Alignment for Interpretable Latent Optimization","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2305.13301","citing_title":"Training Diffusion Models with Reinforcement Learning","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06779","citing_title":"VASR: Variance-Aware Systematic Resampling for Reward-Guided Diffusion","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17397","citing_title":"Speculative Decoding for Autoregressive Video Generation","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SPJUXULNN4JE4UQRFHHXZSCSAY","json":"https://pith.science/pith/SPJUXULNN4JE4UQRFHHXZSCSAY.json","graph_json":"https://pith.science/api/pith-number/SPJUXULNN4JE4UQRFHHXZSCSAY/graph.json","events_json":"https://pith.science/api/pith-number/SPJUXULNN4JE4UQRFHHXZSCSAY/events.json","paper":"https://pith.science/paper/SPJUXULN"},"agent_actions":{"view_html":"https://pith.science/pith/SPJUXULNN4JE4UQRFHHXZSCSAY","download_json":"https://pith.science/pith/SPJUXULNN4JE4UQRFHHXZSCSAY.json","view_paper":"https://pith.science/paper/SPJUXULN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.05977&json=true","fetch_graph":"https://pith.science/api/pith-number/SPJUXULNN4JE4UQRFHHXZSCSAY/graph.json","fetch_events":"https://pith.science/api/pith-number/SPJUXULNN4JE4UQRFHHXZSCSAY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SPJUXULNN4JE4UQRFHHXZSCSAY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SPJUXULNN4JE4UQRFHHXZSCSAY/action/storage_attestation","attest_author":"https://pith.science/pith/SPJUXULNN4JE4UQRFHHXZSCSAY/action/author_attestation","sign_citation":"https://pith.science/pith/SPJUXULNN4JE4UQRFHHXZSCSAY/action/citation_signature","submit_replication":"https://pith.science/pith/SPJUXULNN4JE4UQRFHHXZSCSAY/action/replication_record"}},"created_at":"2026-07-05T07:28:27.841723+00:00","updated_at":"2026-07-05T07:28:27.841723+00:00"}