{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:HKCEZK66GYSVJHWR6PV26EC3MB","short_pith_number":"pith:HKCEZK66","canonical_record":{"source":{"id":"2604.18530","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2026-04-20T17:26:00Z","cross_cats_sorted":[],"title_canon_sha256":"ff665f7e1cb0a558f475abc79644fdadc3f2483e62ea5a6746f0621f18c857e2","abstract_canon_sha256":"e4bf4d3f31f43abc77765d6a6fd84b6f8c25a7eb278d302b1b6d86f25776f9ad"},"schema_version":"1.0"},"canonical_sha256":"3a844cabde3625549ed1f3ebaf105b6074329d4a593e2d39d8452f3ae09b7201","source":{"kind":"arxiv","id":"2604.18530","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.18530","created_at":"2026-05-28T01:04:40Z"},{"alias_kind":"arxiv_version","alias_value":"2604.18530v2","created_at":"2026-05-28T01:04:40Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.18530","created_at":"2026-05-28T01:04:40Z"},{"alias_kind":"pith_short_12","alias_value":"HKCEZK66GYSV","created_at":"2026-05-28T01:04:40Z"},{"alias_kind":"pith_short_16","alias_value":"HKCEZK66GYSVJHWR","created_at":"2026-05-28T01:04:40Z"},{"alias_kind":"pith_short_8","alias_value":"HKCEZK66","created_at":"2026-05-28T01:04:40Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:HKCEZK66GYSVJHWR6PV26EC3MB","target":"record","payload":{"canonical_record":{"source":{"id":"2604.18530","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2026-04-20T17:26:00Z","cross_cats_sorted":[],"title_canon_sha256":"ff665f7e1cb0a558f475abc79644fdadc3f2483e62ea5a6746f0621f18c857e2","abstract_canon_sha256":"e4bf4d3f31f43abc77765d6a6fd84b6f8c25a7eb278d302b1b6d86f25776f9ad"},"schema_version":"1.0"},"canonical_sha256":"3a844cabde3625549ed1f3ebaf105b6074329d4a593e2d39d8452f3ae09b7201","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-28T01:04:40.487851Z","signature_b64":"8NG/shCRgQs7XHBCanNCxx5k5Ss3Ooyb2PFPAkVMfdU6DU0YNSngXgubEmS8Zbv3TqZNdwaj6t/ejWlw9BiTCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3a844cabde3625549ed1f3ebaf105b6074329d4a593e2d39d8452f3ae09b7201","last_reissued_at":"2026-05-28T01:04:40.487248Z","signature_status":"signed_v1","first_computed_at":"2026-05-28T01:04:40.487248Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2604.18530","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-28T01:04:40Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"yhvm4GAsEwdP502EQ+0H/KBpYZTV4mMoPlyk5AyhXtZvZ0IdRcm3ZhiDRspESWe46zk+6biM6LcflVF+VVbTBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T07:19:32.445100Z"},"content_sha256":"5d65bd945ec508ec27de49a91b91a6cedb7f9123eadc370213c9852c828f641e","schema_version":"1.0","event_id":"sha256:5d65bd945ec508ec27de49a91b91a6cedb7f9123eadc370213c9852c828f641e"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:HKCEZK66GYSVJHWR6PV26EC3MB","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"OGER: A Robust Offline-Guided Exploration Reward for Hybrid Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"The OGER framework improves LLM reasoning by integrating offline guidance with an entropy-based exploration reward in hybrid reinforcement learning.","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Chang Jin, Derek F. Wong, Mingzhou Xu, Min Zhang, Qiang Wang, Xinyu Ma, Xuebo Liu","submitted_at":"2026-04-20T17:26:00Z","abstract_excerpt":"Recent advancements in Reinforcement Learning with Verifiable Rewards (RLVR) have significantly improved Large Language Model (LLM) reasoning, yet models often struggle to explore novel trajectories beyond their initial policy distribution. While offline teacher guidance and entropy-driven strategies have been proposed to address this, they often lack deep integration or are constrained by the model's inherent capacity. In this paper, we propose OGER (Offline-Guided Exploration Reward), a novel framework that unifies offline teacher guidance and online reinforcement learning through a speciali"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"OGER significantly outperforms competitive baselines, achieving substantial gains in mathematical reasoning while maintaining robust generalization to out-of-domain tasks.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the entropy-aware reward modulation, when combined with multi-teacher offline guidance, reliably incentivizes useful exploration rather than noise or overfitting to the offline dataset.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"OGER adds an auxiliary exploration reward built from offline trajectories and model entropy to hybrid RL training, yielding gains on math reasoning benchmarks and out-of-domain generalization.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"The OGER framework improves LLM reasoning by integrating offline guidance with an entropy-based exploration reward in hybrid reinforcement learning.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"4b121037efbce4c3575d0627a740001bf9fa3791d5611348870f82046d4c940c"},"source":{"id":"2604.18530","kind":"arxiv","version":2},"verdict":{"id":"81a91c3c-9e31-4a35-8adf-62231c341b3c","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-10T04:26:14.828634Z","strongest_claim":"OGER significantly outperforms competitive baselines, achieving substantial gains in mathematical reasoning while maintaining robust generalization to out-of-domain tasks.","one_line_summary":"OGER adds an auxiliary exploration reward built from offline trajectories and model entropy to hybrid RL training, yielding gains on math reasoning benchmarks and out-of-domain generalization.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the entropy-aware reward modulation, when combined with multi-teacher offline guidance, reliably incentivizes useful exploration rather than noise or overfitting to the offline dataset.","pith_extraction_headline":"The OGER framework improves LLM reasoning by integrating offline guidance with an entropy-based exploration reward in hybrid reinforcement learning."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.18530/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"doi_compliance","ran_at":"2026-05-20T03:54:25.114262Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"fd0573d27b44cbd690b5b4f34adeb5ddc630d841936a6c123b060921fbc81ed8"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"81a91c3c-9e31-4a35-8adf-62231c341b3c"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-28T01:04:40Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ahv5nPhISGazzeBy3abQpGzxiQmHPsS0Hc/pVEPJSF0qSxoCXmRPlU4aAe70UlnUY9Y/WxdGuVXYYX9IxaH7Dw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T07:19:32.445612Z"},"content_sha256":"ecd860a35daf5ce726af442d46f7c382bf9bc4491c94128bdb5bc43959011221","schema_version":"1.0","event_id":"sha256:ecd860a35daf5ce726af442d46f7c382bf9bc4491c94128bdb5bc43959011221"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/HKCEZK66GYSVJHWR6PV26EC3MB/bundle.json","state_url":"https://pith.science/pith/HKCEZK66GYSVJHWR6PV26EC3MB/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/HKCEZK66GYSVJHWR6PV26EC3MB/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T07:19:32Z","links":{"resolver":"https://pith.science/pith/HKCEZK66GYSVJHWR6PV26EC3MB","bundle":"https://pith.science/pith/HKCEZK66GYSVJHWR6PV26EC3MB/bundle.json","state":"https://pith.science/pith/HKCEZK66GYSVJHWR6PV26EC3MB/state.json","well_known_bundle":"https://pith.science/.well-known/pith/HKCEZK66GYSVJHWR6PV26EC3MB/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:HKCEZK66GYSVJHWR6PV26EC3MB","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"e4bf4d3f31f43abc77765d6a6fd84b6f8c25a7eb278d302b1b6d86f25776f9ad","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2026-04-20T17:26:00Z","title_canon_sha256":"ff665f7e1cb0a558f475abc79644fdadc3f2483e62ea5a6746f0621f18c857e2"},"schema_version":"1.0","source":{"id":"2604.18530","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.18530","created_at":"2026-05-28T01:04:40Z"},{"alias_kind":"arxiv_version","alias_value":"2604.18530v2","created_at":"2026-05-28T01:04:40Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.18530","created_at":"2026-05-28T01:04:40Z"},{"alias_kind":"pith_short_12","alias_value":"HKCEZK66GYSV","created_at":"2026-05-28T01:04:40Z"},{"alias_kind":"pith_short_16","alias_value":"HKCEZK66GYSVJHWR","created_at":"2026-05-28T01:04:40Z"},{"alias_kind":"pith_short_8","alias_value":"HKCEZK66","created_at":"2026-05-28T01:04:40Z"}],"graph_snapshots":[{"event_id":"sha256:ecd860a35daf5ce726af442d46f7c382bf9bc4491c94128bdb5bc43959011221","target":"graph","created_at":"2026-05-28T01:04:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"OGER significantly outperforms competitive baselines, achieving substantial gains in mathematical reasoning while maintaining robust generalization to out-of-domain tasks."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That the entropy-aware reward modulation, when combined with multi-teacher offline guidance, reliably incentivizes useful exploration rather than noise or overfitting to the offline dataset."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"OGER adds an auxiliary exploration reward built from offline trajectories and model entropy to hybrid RL training, yielding gains on math reasoning benchmarks and out-of-domain generalization."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"The OGER framework improves LLM reasoning by integrating offline guidance with an entropy-based exploration reward in hybrid reinforcement learning."}],"snapshot_sha256":"4b121037efbce4c3575d0627a740001bf9fa3791d5611348870f82046d4c940c"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[{"findings_count":0,"name":"doi_compliance","ran_at":"2026-05-20T03:54:25.114262Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2604.18530/integrity.json","findings":[],"snapshot_sha256":"fd0573d27b44cbd690b5b4f34adeb5ddc630d841936a6c123b060921fbc81ed8","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Recent advancements in Reinforcement Learning with Verifiable Rewards (RLVR) have significantly improved Large Language Model (LLM) reasoning, yet models often struggle to explore novel trajectories beyond their initial policy distribution. While offline teacher guidance and entropy-driven strategies have been proposed to address this, they often lack deep integration or are constrained by the model's inherent capacity. In this paper, we propose OGER (Offline-Guided Exploration Reward), a novel framework that unifies offline teacher guidance and online reinforcement learning through a speciali","authors_text":"Chang Jin, Derek F. Wong, Mingzhou Xu, Min Zhang, Qiang Wang, Xinyu Ma, Xuebo Liu","cross_cats":[],"headline":"The OGER framework improves LLM reasoning by integrating offline guidance with an entropy-based exploration reward in hybrid reinforcement learning.","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2026-04-20T17:26:00Z","title":"OGER: A Robust Offline-Guided Exploration Reward for Hybrid Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2604.18530","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-10T04:26:14.828634Z","id":"81a91c3c-9e31-4a35-8adf-62231c341b3c","model_set":{"reader":"grok-4.3"},"one_line_summary":"OGER adds an auxiliary exploration reward built from offline trajectories and model entropy to hybrid RL training, yielding gains on math reasoning benchmarks and out-of-domain generalization.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"The OGER framework improves LLM reasoning by integrating offline guidance with an entropy-based exploration reward in hybrid reinforcement learning.","strongest_claim":"OGER significantly outperforms competitive baselines, achieving substantial gains in mathematical reasoning while maintaining robust generalization to out-of-domain tasks.","weakest_assumption":"That the entropy-aware reward modulation, when combined with multi-teacher offline guidance, reliably incentivizes useful exploration rather than noise or overfitting to the offline dataset."}},"verdict_id":"81a91c3c-9e31-4a35-8adf-62231c341b3c"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:5d65bd945ec508ec27de49a91b91a6cedb7f9123eadc370213c9852c828f641e","target":"record","created_at":"2026-05-28T01:04:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"e4bf4d3f31f43abc77765d6a6fd84b6f8c25a7eb278d302b1b6d86f25776f9ad","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2026-04-20T17:26:00Z","title_canon_sha256":"ff665f7e1cb0a558f475abc79644fdadc3f2483e62ea5a6746f0621f18c857e2"},"schema_version":"1.0","source":{"id":"2604.18530","kind":"arxiv","version":2}},"canonical_sha256":"3a844cabde3625549ed1f3ebaf105b6074329d4a593e2d39d8452f3ae09b7201","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3a844cabde3625549ed1f3ebaf105b6074329d4a593e2d39d8452f3ae09b7201","first_computed_at":"2026-05-28T01:04:40.487248Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-28T01:04:40.487248Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"8NG/shCRgQs7XHBCanNCxx5k5Ss3Ooyb2PFPAkVMfdU6DU0YNSngXgubEmS8Zbv3TqZNdwaj6t/ejWlw9BiTCw==","signature_status":"signed_v1","signed_at":"2026-05-28T01:04:40.487851Z","signed_message":"canonical_sha256_bytes"},"source_id":"2604.18530","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:5d65bd945ec508ec27de49a91b91a6cedb7f9123eadc370213c9852c828f641e","sha256:ecd860a35daf5ce726af442d46f7c382bf9bc4491c94128bdb5bc43959011221"],"state_sha256":"f32a778f6cbd2dfa00084480b92b2b8ced34f8068d1933c281e37cd73291713f"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"rj7F6xXkL6I1O4bko5fMs0niJFJgNCQOm7GurQJr1ZxsSC2A8OMHzMamZ2xYnC2zYbFa708DBxNHgBvIHO5ZAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T07:19:32.448573Z","bundle_sha256":"fb5ef4e16572d7d303519cf3855aea6e6f8591467d64bd75b6c407505acfcea3"}}