{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BWR27SCWVSRVAZD5QUNICHWCMU","short_pith_number":"pith:BWR27SCW","schema_version":"1.0","canonical_sha256":"0da3afc856aca350647d851a811ec26505974112f129f1bf1d305b9d9587d4e2","source":{"kind":"arxiv","id":"2506.15421","version":1},"attestation_state":"computed","paper":{"title":"Reward Models in Deep Reinforcement Learning: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chen-Xiao Gao, De-Chuan Zhan, Le Gan, Rui Yu, ShengHua Wan, Yucen Wang, Zongzhang Zhang","submitted_at":"2025-06-18T12:46:39Z","abstract_excerpt":"In reinforcement learning (RL), agents continually interact with the environment and use the feedback to refine their behavior. To guide policy optimization, reward models are introduced as proxies of the desired objectives, such that when the agent maximizes the accumulated reward, it also fulfills the task designer's intentions. Recently, significant attention from both academic and industrial researchers has focused on developing reward models that not only align closely with the true objectives but also facilitate policy optimization. In this survey, we provide a comprehensive review of re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.15421","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-06-18T12:46:39Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8795543cac56b8cff78a3c1f2c82c2ddd19f82cfce4018a22dda2fa3c8e372d8","abstract_canon_sha256":"a62adc370da5746205d7070053ddeceacb2eb5a33d38068d8c777313cbc553e1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:23:42.967030Z","signature_b64":"WupEhwzL0fvSMa5r820hJrkCUfDPF8h5mx+s78WHCCV47AV2lZ2zoE7Yx1w443cRTHA38VDCKO52hte+1b8ABg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0da3afc856aca350647d851a811ec26505974112f129f1bf1d305b9d9587d4e2","last_reissued_at":"2026-07-05T11:23:42.966540Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:23:42.966540Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reward Models in Deep Reinforcement Learning: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chen-Xiao Gao, De-Chuan Zhan, Le Gan, Rui Yu, ShengHua Wan, Yucen Wang, Zongzhang Zhang","submitted_at":"2025-06-18T12:46:39Z","abstract_excerpt":"In reinforcement learning (RL), agents continually interact with the environment and use the feedback to refine their behavior. To guide policy optimization, reward models are introduced as proxies of the desired objectives, such that when the agent maximizes the accumulated reward, it also fulfills the task designer's intentions. Recently, significant attention from both academic and industrial researchers has focused on developing reward models that not only align closely with the true objectives but also facilitate policy optimization. In this survey, we provide a comprehensive review of re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.15421","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.15421/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.15421","created_at":"2026-07-05T11:23:42.966602+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.15421v1","created_at":"2026-07-05T11:23:42.966602+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.15421","created_at":"2026-07-05T11:23:42.966602+00:00"},{"alias_kind":"pith_short_12","alias_value":"BWR27SCWVSRV","created_at":"2026-07-05T11:23:42.966602+00:00"},{"alias_kind":"pith_short_16","alias_value":"BWR27SCWVSRVAZD5","created_at":"2026-07-05T11:23:42.966602+00:00"},{"alias_kind":"pith_short_8","alias_value":"BWR27SCW","created_at":"2026-07-05T11:23:42.966602+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26930","citing_title":"PortraitGen: Exemplar-Driven GRPO with Dual-Reward Guidance for Photorealistic Portrait Generation","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09925","citing_title":"AudioProcessBench: Benchmark for Identifying Process Errors in Audio-Grounded Reasoning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13276","citing_title":"D-VLA: A High-Concurrency Distributed Asynchronous Reinforcement Learning Framework for Vision-Language-Action Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13276","citing_title":"D-VLA: A High-Concurrency Distributed Asynchronous Reinforcement Learning Framework for Vision-Language-Action Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15757","citing_title":"Multi-objective Reinforcement Learning With Augmented States Requires Rewards After Deployment","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20627","citing_title":"Occupancy Reward Shaping: Improving Credit Assignment for Offline Goal-Conditioned Reinforcement Learning","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BWR27SCWVSRVAZD5QUNICHWCMU","json":"https://pith.science/pith/BWR27SCWVSRVAZD5QUNICHWCMU.json","graph_json":"https://pith.science/api/pith-number/BWR27SCWVSRVAZD5QUNICHWCMU/graph.json","events_json":"https://pith.science/api/pith-number/BWR27SCWVSRVAZD5QUNICHWCMU/events.json","paper":"https://pith.science/paper/BWR27SCW"},"agent_actions":{"view_html":"https://pith.science/pith/BWR27SCWVSRVAZD5QUNICHWCMU","download_json":"https://pith.science/pith/BWR27SCWVSRVAZD5QUNICHWCMU.json","view_paper":"https://pith.science/paper/BWR27SCW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.15421&json=true","fetch_graph":"https://pith.science/api/pith-number/BWR27SCWVSRVAZD5QUNICHWCMU/graph.json","fetch_events":"https://pith.science/api/pith-number/BWR27SCWVSRVAZD5QUNICHWCMU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BWR27SCWVSRVAZD5QUNICHWCMU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BWR27SCWVSRVAZD5QUNICHWCMU/action/storage_attestation","attest_author":"https://pith.science/pith/BWR27SCWVSRVAZD5QUNICHWCMU/action/author_attestation","sign_citation":"https://pith.science/pith/BWR27SCWVSRVAZD5QUNICHWCMU/action/citation_signature","submit_replication":"https://pith.science/pith/BWR27SCWVSRVAZD5QUNICHWCMU/action/replication_record"}},"created_at":"2026-07-05T11:23:42.966602+00:00","updated_at":"2026-07-05T11:23:42.966602+00:00"}