{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:R6VFQURKQYSBR2X5SHTS6JDJJK","short_pith_number":"pith:R6VFQURK","canonical_record":{"source":{"id":"2505.23585","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-29T15:58:04Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"585c60f21382c0e814e8efe97e06de00aa8d6c9592975495bc98e5944f155ca9","abstract_canon_sha256":"5cf6ceb2f728a259142456ddfc712569944f88795ca11cb00ea6bb6b9a840817"},"schema_version":"1.0"},"canonical_sha256":"8faa58522a862418eafd91e72f24694a8f1e60cc7775d94e18bae6312072fa64","source":{"kind":"arxiv","id":"2505.23585","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.23585","created_at":"2026-07-05T11:15:21Z"},{"alias_kind":"arxiv_version","alias_value":"2505.23585v2","created_at":"2026-07-05T11:15:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23585","created_at":"2026-07-05T11:15:21Z"},{"alias_kind":"pith_short_12","alias_value":"R6VFQURKQYSB","created_at":"2026-07-05T11:15:21Z"},{"alias_kind":"pith_short_16","alias_value":"R6VFQURKQYSBR2X5","created_at":"2026-07-05T11:15:21Z"},{"alias_kind":"pith_short_8","alias_value":"R6VFQURK","created_at":"2026-07-05T11:15:21Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:R6VFQURKQYSBR2X5SHTS6JDJJK","target":"record","payload":{"canonical_record":{"source":{"id":"2505.23585","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-29T15:58:04Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"585c60f21382c0e814e8efe97e06de00aa8d6c9592975495bc98e5944f155ca9","abstract_canon_sha256":"5cf6ceb2f728a259142456ddfc712569944f88795ca11cb00ea6bb6b9a840817"},"schema_version":"1.0"},"canonical_sha256":"8faa58522a862418eafd91e72f24694a8f1e60cc7775d94e18bae6312072fa64","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:21.547283Z","signature_b64":"6Z2DWed7JqaoIju2gtzlG4eEzzm2g35GjrJ6Oa/uYzoe/LJ+Yshia8cxSjRbNEkGT/NtC9T2yi65EwBtV1ZIAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8faa58522a862418eafd91e72f24694a8f1e60cc7775d94e18bae6312072fa64","last_reissued_at":"2026-07-05T11:15:21.546816Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:21.546816Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2505.23585","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:15:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"KSn5lQMFB41t4ehj9rXj6+zO90BXYg1PlR6O7rn5efFYBol2sWkHbWaQwGMqWXWT6thzKqWV+Q+9UA7yVlSCDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T02:14:11.817375Z"},"content_sha256":"b3cfb23692e0b8731471efccd2a46eda35a45cf57b95267b1ef24091496a99e3","schema_version":"1.0","event_id":"sha256:b3cfb23692e0b8731471efccd2a46eda35a45cf57b95267b1ef24091496a99e3"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:R6VFQURKQYSBR2X5SHTS6JDJJK","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"On-Policy RL with Optimal Reward Baseline","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Furu Wei, Li Dong, Shaohan Huang, Xun Wu, Yaru Hao, Zewen Chi","submitted_at":"2025-05-29T15:58:04Z","abstract_excerpt":"Reinforcement learning algorithms are fundamental to align large language models with human preferences and to enhance their reasoning capabilities. However, current reinforcement learning algorithms often suffer from training instability due to loose on-policy constraints and computational inefficiency due to auxiliary models. In this work, we propose On-Policy RL with Optimal reward baseline (OPO), a novel and simplified reinforcement learning algorithm designed to address these challenges. OPO emphasizes the importance of exact on-policy training, which empirically stabilizes the training p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23585","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23585/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:15:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"bP2cuboqMViHa8PLF5NWwSXC744Zqk9QMecEOnH5W82W1cukWRyBUWivFL6wZsHWGAaFoXTW3COqw18ocggfDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T02:14:11.817959Z"},"content_sha256":"bb18b11e09cb0e94554f2f2cdb15d31f859f8e71b181113db33b0ade6398eca7","schema_version":"1.0","event_id":"sha256:bb18b11e09cb0e94554f2f2cdb15d31f859f8e71b181113db33b0ade6398eca7"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/R6VFQURKQYSBR2X5SHTS6JDJJK/bundle.json","state_url":"https://pith.science/pith/R6VFQURKQYSBR2X5SHTS6JDJJK/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/R6VFQURKQYSBR2X5SHTS6JDJJK/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T02:14:11Z","links":{"resolver":"https://pith.science/pith/R6VFQURKQYSBR2X5SHTS6JDJJK","bundle":"https://pith.science/pith/R6VFQURKQYSBR2X5SHTS6JDJJK/bundle.json","state":"https://pith.science/pith/R6VFQURKQYSBR2X5SHTS6JDJJK/state.json","well_known_bundle":"https://pith.science/.well-known/pith/R6VFQURKQYSBR2X5SHTS6JDJJK/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:R6VFQURKQYSBR2X5SHTS6JDJJK","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5cf6ceb2f728a259142456ddfc712569944f88795ca11cb00ea6bb6b9a840817","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-29T15:58:04Z","title_canon_sha256":"585c60f21382c0e814e8efe97e06de00aa8d6c9592975495bc98e5944f155ca9"},"schema_version":"1.0","source":{"id":"2505.23585","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.23585","created_at":"2026-07-05T11:15:21Z"},{"alias_kind":"arxiv_version","alias_value":"2505.23585v2","created_at":"2026-07-05T11:15:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23585","created_at":"2026-07-05T11:15:21Z"},{"alias_kind":"pith_short_12","alias_value":"R6VFQURKQYSB","created_at":"2026-07-05T11:15:21Z"},{"alias_kind":"pith_short_16","alias_value":"R6VFQURKQYSBR2X5","created_at":"2026-07-05T11:15:21Z"},{"alias_kind":"pith_short_8","alias_value":"R6VFQURK","created_at":"2026-07-05T11:15:21Z"}],"graph_snapshots":[{"event_id":"sha256:bb18b11e09cb0e94554f2f2cdb15d31f859f8e71b181113db33b0ade6398eca7","target":"graph","created_at":"2026-07-05T11:15:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.23585/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning algorithms are fundamental to align large language models with human preferences and to enhance their reasoning capabilities. However, current reinforcement learning algorithms often suffer from training instability due to loose on-policy constraints and computational inefficiency due to auxiliary models. In this work, we propose On-Policy RL with Optimal reward baseline (OPO), a novel and simplified reinforcement learning algorithm designed to address these challenges. OPO emphasizes the importance of exact on-policy training, which empirically stabilizes the training p","authors_text":"Furu Wei, Li Dong, Shaohan Huang, Xun Wu, Yaru Hao, Zewen Chi","cross_cats":["cs.CL"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-29T15:58:04Z","title":"On-Policy RL with Optimal Reward Baseline"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23585","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:b3cfb23692e0b8731471efccd2a46eda35a45cf57b95267b1ef24091496a99e3","target":"record","created_at":"2026-07-05T11:15:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5cf6ceb2f728a259142456ddfc712569944f88795ca11cb00ea6bb6b9a840817","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-29T15:58:04Z","title_canon_sha256":"585c60f21382c0e814e8efe97e06de00aa8d6c9592975495bc98e5944f155ca9"},"schema_version":"1.0","source":{"id":"2505.23585","kind":"arxiv","version":2}},"canonical_sha256":"8faa58522a862418eafd91e72f24694a8f1e60cc7775d94e18bae6312072fa64","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"8faa58522a862418eafd91e72f24694a8f1e60cc7775d94e18bae6312072fa64","first_computed_at":"2026-07-05T11:15:21.546816Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:15:21.546816Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"6Z2DWed7JqaoIju2gtzlG4eEzzm2g35GjrJ6Oa/uYzoe/LJ+Yshia8cxSjRbNEkGT/NtC9T2yi65EwBtV1ZIAw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:15:21.547283Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.23585","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:b3cfb23692e0b8731471efccd2a46eda35a45cf57b95267b1ef24091496a99e3","sha256:bb18b11e09cb0e94554f2f2cdb15d31f859f8e71b181113db33b0ade6398eca7"],"state_sha256":"717c3ab76c6ac01889eadab6cbb940f49479c97f6677618b16fa870121619ae5"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ene5wVVeTP+S2YiNEonKxXpAwux3Nmkpw1eznGuEpMuicmxEZuwdSNs9sBU1Q5kK9KibdieenWFPCiG6nIS8Aw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T02:14:11.823282Z","bundle_sha256":"2871e3632d64701088939af4274744b75c8d38348d170b891f2f140f9fea4fe6"}}