{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LM6PQ2UIFVWSUCPPD27NTVMN7J","short_pith_number":"pith:LM6PQ2UI","schema_version":"1.0","canonical_sha256":"5b3cf86a882d6d2a09ef1ebed9d58dfa6cc9fdb0cc5e3dd21d4f3ff52b0a4008","source":{"kind":"arxiv","id":"2505.17910","version":1},"attestation_state":"computed","paper":{"title":"DiffusionReward: Enhancing Blind Face Restoration through Reward Feedback Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bin Wu, Wei Wang, Yahui Liu, Yao Zhao, Zixiang Li","submitted_at":"2025-05-23T13:53:23Z","abstract_excerpt":"Reward Feedback Learning (ReFL) has recently shown great potential in aligning model outputs with human preferences across various generative tasks. In this work, we introduce a ReFL framework, named DiffusionReward, to the Blind Face Restoration task for the first time. DiffusionReward effectively overcomes the limitations of diffusion-based methods, which often fail to generate realistic facial details and exhibit poor identity consistency. The core of our framework is the Face Reward Model (FRM), which is trained using carefully annotated data. It provides feedback signals that play a pivot"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.17910","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-23T13:53:23Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9fd9dcfcea01d148b889882376ea3bcb3ce507d78d4d0a374c75549dfa5bcf2d","abstract_canon_sha256":"d3724cb502f11322bcdfea22cce9854712248637eccc5ef22ae89b6a6296feef"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:33.700468Z","signature_b64":"wvqN/NqvnBvDSobim99irZvhteAPVRUb6sJMh5p8kCwrFv0fSdwGgLWHCoqi0kPqk3KDxe1NfuGqKxsWEbAKBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5b3cf86a882d6d2a09ef1ebed9d58dfa6cc9fdb0cc5e3dd21d4f3ff52b0a4008","last_reissued_at":"2026-07-05T11:08:33.699975Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:33.699975Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DiffusionReward: Enhancing Blind Face Restoration through Reward Feedback Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bin Wu, Wei Wang, Yahui Liu, Yao Zhao, Zixiang Li","submitted_at":"2025-05-23T13:53:23Z","abstract_excerpt":"Reward Feedback Learning (ReFL) has recently shown great potential in aligning model outputs with human preferences across various generative tasks. In this work, we introduce a ReFL framework, named DiffusionReward, to the Blind Face Restoration task for the first time. DiffusionReward effectively overcomes the limitations of diffusion-based methods, which often fail to generate realistic facial details and exhibit poor identity consistency. The core of our framework is the Face Reward Model (FRM), which is trained using carefully annotated data. It provides feedback signals that play a pivot"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.17910","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.17910/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.17910","created_at":"2026-07-05T11:08:33.700040+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.17910v1","created_at":"2026-07-05T11:08:33.700040+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.17910","created_at":"2026-07-05T11:08:33.700040+00:00"},{"alias_kind":"pith_short_12","alias_value":"LM6PQ2UIFVWS","created_at":"2026-07-05T11:08:33.700040+00:00"},{"alias_kind":"pith_short_16","alias_value":"LM6PQ2UIFVWSUCPP","created_at":"2026-07-05T11:08:33.700040+00:00"},{"alias_kind":"pith_short_8","alias_value":"LM6PQ2UI","created_at":"2026-07-05T11:08:33.700040+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09711","citing_title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","ref_index":294,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13602","citing_title":"Reward Hacking in the Era of Large Models: Mechanisms, Emergent Misalignment, Challenges","ref_index":196,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LM6PQ2UIFVWSUCPPD27NTVMN7J","json":"https://pith.science/pith/LM6PQ2UIFVWSUCPPD27NTVMN7J.json","graph_json":"https://pith.science/api/pith-number/LM6PQ2UIFVWSUCPPD27NTVMN7J/graph.json","events_json":"https://pith.science/api/pith-number/LM6PQ2UIFVWSUCPPD27NTVMN7J/events.json","paper":"https://pith.science/paper/LM6PQ2UI"},"agent_actions":{"view_html":"https://pith.science/pith/LM6PQ2UIFVWSUCPPD27NTVMN7J","download_json":"https://pith.science/pith/LM6PQ2UIFVWSUCPPD27NTVMN7J.json","view_paper":"https://pith.science/paper/LM6PQ2UI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.17910&json=true","fetch_graph":"https://pith.science/api/pith-number/LM6PQ2UIFVWSUCPPD27NTVMN7J/graph.json","fetch_events":"https://pith.science/api/pith-number/LM6PQ2UIFVWSUCPPD27NTVMN7J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LM6PQ2UIFVWSUCPPD27NTVMN7J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LM6PQ2UIFVWSUCPPD27NTVMN7J/action/storage_attestation","attest_author":"https://pith.science/pith/LM6PQ2UIFVWSUCPPD27NTVMN7J/action/author_attestation","sign_citation":"https://pith.science/pith/LM6PQ2UIFVWSUCPPD27NTVMN7J/action/citation_signature","submit_replication":"https://pith.science/pith/LM6PQ2UIFVWSUCPPD27NTVMN7J/action/replication_record"}},"created_at":"2026-07-05T11:08:33.700040+00:00","updated_at":"2026-07-05T11:08:33.700040+00:00"}