{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K3N7M2XUQHMSXKFHGC5Y77DMUS","short_pith_number":"pith:K3N7M2XU","schema_version":"1.0","canonical_sha256":"56dbf66af481d92ba8a730bb8ffc6ca4a73738f50ae483499e5d7f2a4d6d1379","source":{"kind":"arxiv","id":"2404.16767","version":4},"attestation_state":"computed","paper":{"title":"REBEL: Reinforcement Learning via Regressing Relative Rewards","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Gokul Swamy, J. Andrew Bagnell, Jason D. Lee, Jonathan D. Chang, Kiant\\'e Brantley, Owen Oertell, Thorsten Joachims, Wenhao Zhan, Wen Sun, Zhaolin Gao","submitted_at":"2024-04-25T17:20:45Z","abstract_excerpt":"While originally developed for continuous control problems, Proximal Policy Optimization (PPO) has emerged as the work-horse of a variety of reinforcement learning (RL) applications, including the fine-tuning of generative models. Unfortunately, PPO requires multiple heuristics to enable stable convergence (e.g. value networks, clipping), and is notorious for its sensitivity to the precise implementation of these components. In response, we take a step back and ask what a minimalist RL algorithm for the era of generative models would look like. We propose REBEL, an algorithm that cleanly reduc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.16767","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-25T17:20:45Z","cross_cats_sorted":["cs.CL","cs.CV"],"title_canon_sha256":"737c3120a86adcfa8c24f66c506e2421a9f92cb9d18414d5d113e2402e7b7c3b","abstract_canon_sha256":"45852ca6e401d899c5ebb0c6cd6a50b79a3f550d61483afcae0d4d8b7a220987"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:46:46.553140Z","signature_b64":"O24u6rOXHKekrRy3B8xsyJjr+qOu1A1B09kZ1CJeuAsp2xbmgKcUJlhi0QHlx1Js4om1qLVqklP/YNulyon2Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"56dbf66af481d92ba8a730bb8ffc6ca4a73738f50ae483499e5d7f2a4d6d1379","last_reissued_at":"2026-07-05T09:46:46.552596Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:46:46.552596Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"REBEL: Reinforcement Learning via Regressing Relative Rewards","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Gokul Swamy, J. Andrew Bagnell, Jason D. Lee, Jonathan D. Chang, Kiant\\'e Brantley, Owen Oertell, Thorsten Joachims, Wenhao Zhan, Wen Sun, Zhaolin Gao","submitted_at":"2024-04-25T17:20:45Z","abstract_excerpt":"While originally developed for continuous control problems, Proximal Policy Optimization (PPO) has emerged as the work-horse of a variety of reinforcement learning (RL) applications, including the fine-tuning of generative models. Unfortunately, PPO requires multiple heuristics to enable stable convergence (e.g. value networks, clipping), and is notorious for its sensitivity to the precise implementation of these components. In response, we take a step back and ask what a minimalist RL algorithm for the era of generative models would look like. We propose REBEL, an algorithm that cleanly reduc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.16767","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.16767/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.16767","created_at":"2026-07-05T09:46:46.552656+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.16767v4","created_at":"2026-07-05T09:46:46.552656+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.16767","created_at":"2026-07-05T09:46:46.552656+00:00"},{"alias_kind":"pith_short_12","alias_value":"K3N7M2XUQHMS","created_at":"2026-07-05T09:46:46.552656+00:00"},{"alias_kind":"pith_short_16","alias_value":"K3N7M2XUQHMSXKFH","created_at":"2026-07-05T09:46:46.552656+00:00"},{"alias_kind":"pith_short_8","alias_value":"K3N7M2XU","created_at":"2026-07-05T09:46:46.552656+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.21060","citing_title":"On the Sample Complexity of Differentially Private Policy Optimization","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06159","citing_title":"Target Policy Optimization","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K3N7M2XUQHMSXKFHGC5Y77DMUS","json":"https://pith.science/pith/K3N7M2XUQHMSXKFHGC5Y77DMUS.json","graph_json":"https://pith.science/api/pith-number/K3N7M2XUQHMSXKFHGC5Y77DMUS/graph.json","events_json":"https://pith.science/api/pith-number/K3N7M2XUQHMSXKFHGC5Y77DMUS/events.json","paper":"https://pith.science/paper/K3N7M2XU"},"agent_actions":{"view_html":"https://pith.science/pith/K3N7M2XUQHMSXKFHGC5Y77DMUS","download_json":"https://pith.science/pith/K3N7M2XUQHMSXKFHGC5Y77DMUS.json","view_paper":"https://pith.science/paper/K3N7M2XU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.16767&json=true","fetch_graph":"https://pith.science/api/pith-number/K3N7M2XUQHMSXKFHGC5Y77DMUS/graph.json","fetch_events":"https://pith.science/api/pith-number/K3N7M2XUQHMSXKFHGC5Y77DMUS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K3N7M2XUQHMSXKFHGC5Y77DMUS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K3N7M2XUQHMSXKFHGC5Y77DMUS/action/storage_attestation","attest_author":"https://pith.science/pith/K3N7M2XUQHMSXKFHGC5Y77DMUS/action/author_attestation","sign_citation":"https://pith.science/pith/K3N7M2XUQHMSXKFHGC5Y77DMUS/action/citation_signature","submit_replication":"https://pith.science/pith/K3N7M2XUQHMSXKFHGC5Y77DMUS/action/replication_record"}},"created_at":"2026-07-05T09:46:46.552656+00:00","updated_at":"2026-07-05T09:46:46.552656+00:00"}