{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DCB2AYWE6XBK2E2D3OD6GLUKWW","short_pith_number":"pith:DCB2AYWE","schema_version":"1.0","canonical_sha256":"1883a062c4f5c2ad1343db87e32e8ab5836267834b0a2d57dd035eb53099f149","source":{"kind":"arxiv","id":"2412.03268","version":1},"attestation_state":"computed","paper":{"title":"RFSR: Improving ISR Diffusion Models via Reward Feedback Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengjian Feng, Dengjie Li, Jie Hu, Lin Ma, Qinwei Lin, Xiaopeng Sun, Yu Gao, Yujie Zhong, Zheng Zhao","submitted_at":"2024-12-04T12:23:17Z","abstract_excerpt":"Generative diffusion models (DM) have been extensively utilized in image super-resolution (ISR). Most of the existing methods adopt the denoising loss from DDPMs for model optimization. We posit that introducing reward feedback learning to finetune the existing models can further improve the quality of the generated images. In this paper, we propose a timestep-aware training strategy with reward feedback learning. Specifically, in the initial denoising stages of ISR diffusion, we apply low-frequency constraints to super-resolution (SR) images to maintain structural stability. In the later deno"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.03268","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-04T12:23:17Z","cross_cats_sorted":[],"title_canon_sha256":"b3b92229afcea84ea8a159afe25ec20651e222c116f8db28e4b2fdfd2db55489","abstract_canon_sha256":"f9d5830cc40cf511b2025c4479f31e2c2376190362e03869f4cc4267f3360f2c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:44:27.160715Z","signature_b64":"g4Hry8wIeRX39sLswRxDpSYrCPIxQ6KeBzl472ueXkSYkMlgSIHIrjChkSZOsNFAmhPh8LeFcd3huhHcCQpwAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1883a062c4f5c2ad1343db87e32e8ab5836267834b0a2d57dd035eb53099f149","last_reissued_at":"2026-07-05T09:44:27.160248Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:44:27.160248Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RFSR: Improving ISR Diffusion Models via Reward Feedback Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengjian Feng, Dengjie Li, Jie Hu, Lin Ma, Qinwei Lin, Xiaopeng Sun, Yu Gao, Yujie Zhong, Zheng Zhao","submitted_at":"2024-12-04T12:23:17Z","abstract_excerpt":"Generative diffusion models (DM) have been extensively utilized in image super-resolution (ISR). Most of the existing methods adopt the denoising loss from DDPMs for model optimization. We posit that introducing reward feedback learning to finetune the existing models can further improve the quality of the generated images. In this paper, we propose a timestep-aware training strategy with reward feedback learning. Specifically, in the initial denoising stages of ISR diffusion, we apply low-frequency constraints to super-resolution (SR) images to maintain structural stability. In the later deno"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.03268","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.03268/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.03268","created_at":"2026-07-05T09:44:27.160307+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.03268v1","created_at":"2026-07-05T09:44:27.160307+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.03268","created_at":"2026-07-05T09:44:27.160307+00:00"},{"alias_kind":"pith_short_12","alias_value":"DCB2AYWE6XBK","created_at":"2026-07-05T09:44:27.160307+00:00"},{"alias_kind":"pith_short_16","alias_value":"DCB2AYWE6XBK2E2D","created_at":"2026-07-05T09:44:27.160307+00:00"},{"alias_kind":"pith_short_8","alias_value":"DCB2AYWE","created_at":"2026-07-05T09:44:27.160307+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09711","citing_title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","ref_index":291,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13602","citing_title":"Reward Hacking in the Era of Large Models: Mechanisms, Emergent Misalignment, Challenges","ref_index":213,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DCB2AYWE6XBK2E2D3OD6GLUKWW","json":"https://pith.science/pith/DCB2AYWE6XBK2E2D3OD6GLUKWW.json","graph_json":"https://pith.science/api/pith-number/DCB2AYWE6XBK2E2D3OD6GLUKWW/graph.json","events_json":"https://pith.science/api/pith-number/DCB2AYWE6XBK2E2D3OD6GLUKWW/events.json","paper":"https://pith.science/paper/DCB2AYWE"},"agent_actions":{"view_html":"https://pith.science/pith/DCB2AYWE6XBK2E2D3OD6GLUKWW","download_json":"https://pith.science/pith/DCB2AYWE6XBK2E2D3OD6GLUKWW.json","view_paper":"https://pith.science/paper/DCB2AYWE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.03268&json=true","fetch_graph":"https://pith.science/api/pith-number/DCB2AYWE6XBK2E2D3OD6GLUKWW/graph.json","fetch_events":"https://pith.science/api/pith-number/DCB2AYWE6XBK2E2D3OD6GLUKWW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DCB2AYWE6XBK2E2D3OD6GLUKWW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DCB2AYWE6XBK2E2D3OD6GLUKWW/action/storage_attestation","attest_author":"https://pith.science/pith/DCB2AYWE6XBK2E2D3OD6GLUKWW/action/author_attestation","sign_citation":"https://pith.science/pith/DCB2AYWE6XBK2E2D3OD6GLUKWW/action/citation_signature","submit_replication":"https://pith.science/pith/DCB2AYWE6XBK2E2D3OD6GLUKWW/action/replication_record"}},"created_at":"2026-07-05T09:44:27.160307+00:00","updated_at":"2026-07-05T09:44:27.160307+00:00"}