{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:BWCXFO5GNG6AXSFCK6YVSDNQG2","short_pith_number":"pith:BWCXFO5G","canonical_record":{"source":{"id":"2507.10628","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-14T08:10:00Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fa2bdc12fbf36674e897298b2517d06c69c7195498b60e31be9ebc963c5dffa9","abstract_canon_sha256":"2a9a34f39ca0decba791a92f9733d4b9047bc2f6194ba1c13879dfc26979a3f2"},"schema_version":"1.0"},"canonical_sha256":"0d8572bba669bc0bc8a257b1590db0369bef7b077bf494f6de4448203094df6d","source":{"kind":"arxiv","id":"2507.10628","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2507.10628","created_at":"2026-07-05T11:38:09Z"},{"alias_kind":"arxiv_version","alias_value":"2507.10628v2","created_at":"2026-07-05T11:38:09Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.10628","created_at":"2026-07-05T11:38:09Z"},{"alias_kind":"pith_short_12","alias_value":"BWCXFO5GNG6A","created_at":"2026-07-05T11:38:09Z"},{"alias_kind":"pith_short_16","alias_value":"BWCXFO5GNG6AXSFC","created_at":"2026-07-05T11:38:09Z"},{"alias_kind":"pith_short_8","alias_value":"BWCXFO5G","created_at":"2026-07-05T11:38:09Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:BWCXFO5GNG6AXSFCK6YVSDNQG2","target":"record","payload":{"canonical_record":{"source":{"id":"2507.10628","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-14T08:10:00Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fa2bdc12fbf36674e897298b2517d06c69c7195498b60e31be9ebc963c5dffa9","abstract_canon_sha256":"2a9a34f39ca0decba791a92f9733d4b9047bc2f6194ba1c13879dfc26979a3f2"},"schema_version":"1.0"},"canonical_sha256":"0d8572bba669bc0bc8a257b1590db0369bef7b077bf494f6de4448203094df6d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:38:09.650540Z","signature_b64":"uZezRo0gYe3nXf+dzK5Ci88DoR90Y1fPho3hwKqzqTEumkEvQ/oGtPktLevHx/BhB+2tB3HnCy0dXL99NXBECg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0d8572bba669bc0bc8a257b1590db0369bef7b077bf494f6de4448203094df6d","last_reissued_at":"2026-07-05T11:38:09.649964Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:38:09.649964Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2507.10628","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:38:09Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"CH8ULqf827CgjgXmnkiJnttc7dNEGl7LnU4+jZ7FED1OIO9ILzGZuXGPLiErtJkmjEP3HruBib5ZMALQ6+NUAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T08:18:56.726506Z"},"content_sha256":"b5a27c4a8faa1b8507d57f1b8533ac69e6ab0efc740a14100c5003592be4a9ce","schema_version":"1.0","event_id":"sha256:b5a27c4a8faa1b8507d57f1b8533ac69e6ab0efc740a14100c5003592be4a9ce"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:BWCXFO5GNG6AXSFCK6YVSDNQG2","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"GHPO: Adaptive Guidance for Stable and Efficient LLM Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Cheng Gong, Dandan Tu, Qingfu Zhang, Ran Chen, Rui Liu, Shoubo Hu, Suiyun Zhang, Xinyu Fu, Yaofang Liu, Ziru Liu","submitted_at":"2025-07-14T08:10:00Z","abstract_excerpt":"Reinforcement Learning with Verifiable Rewards (RLVR) has recently emerged as a powerful paradigm for facilitating the self-improvement of large language models (LLMs), particularly in the domain of complex reasoning tasks. However, prevailing on-policy RL methods often contend with significant training instability and inefficiency. This is primarily due to a capacity-difficulty mismatch, where the complexity of training data frequently outpaces the model's current capabilities, leading to critically sparse reward signals and stalled learning progress. This challenge is particularly acute for "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.10628","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.10628/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:38:09Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Kc4ztKzDO07PqJUPzlUe/Rg0aMkN2QjZHcnD87ckBbPBHD2GBaab0Hj2TKtsndf/aCejXkXyGdBHJ+49zyPdDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T08:18:56.727025Z"},"content_sha256":"7bfaacbbdecb3df973a034df16d6bc92f923a4ebc2a4e7ff0cc48edf0f0f9da3","schema_version":"1.0","event_id":"sha256:7bfaacbbdecb3df973a034df16d6bc92f923a4ebc2a4e7ff0cc48edf0f0f9da3"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/BWCXFO5GNG6AXSFCK6YVSDNQG2/bundle.json","state_url":"https://pith.science/pith/BWCXFO5GNG6AXSFCK6YVSDNQG2/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/BWCXFO5GNG6AXSFCK6YVSDNQG2/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T08:18:56Z","links":{"resolver":"https://pith.science/pith/BWCXFO5GNG6AXSFCK6YVSDNQG2","bundle":"https://pith.science/pith/BWCXFO5GNG6AXSFCK6YVSDNQG2/bundle.json","state":"https://pith.science/pith/BWCXFO5GNG6AXSFCK6YVSDNQG2/state.json","well_known_bundle":"https://pith.science/.well-known/pith/BWCXFO5GNG6AXSFCK6YVSDNQG2/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:BWCXFO5GNG6AXSFCK6YVSDNQG2","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"2a9a34f39ca0decba791a92f9733d4b9047bc2f6194ba1c13879dfc26979a3f2","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-14T08:10:00Z","title_canon_sha256":"fa2bdc12fbf36674e897298b2517d06c69c7195498b60e31be9ebc963c5dffa9"},"schema_version":"1.0","source":{"id":"2507.10628","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2507.10628","created_at":"2026-07-05T11:38:09Z"},{"alias_kind":"arxiv_version","alias_value":"2507.10628v2","created_at":"2026-07-05T11:38:09Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.10628","created_at":"2026-07-05T11:38:09Z"},{"alias_kind":"pith_short_12","alias_value":"BWCXFO5GNG6A","created_at":"2026-07-05T11:38:09Z"},{"alias_kind":"pith_short_16","alias_value":"BWCXFO5GNG6AXSFC","created_at":"2026-07-05T11:38:09Z"},{"alias_kind":"pith_short_8","alias_value":"BWCXFO5G","created_at":"2026-07-05T11:38:09Z"}],"graph_snapshots":[{"event_id":"sha256:7bfaacbbdecb3df973a034df16d6bc92f923a4ebc2a4e7ff0cc48edf0f0f9da3","target":"graph","created_at":"2026-07-05T11:38:09Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2507.10628/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning with Verifiable Rewards (RLVR) has recently emerged as a powerful paradigm for facilitating the self-improvement of large language models (LLMs), particularly in the domain of complex reasoning tasks. However, prevailing on-policy RL methods often contend with significant training instability and inefficiency. This is primarily due to a capacity-difficulty mismatch, where the complexity of training data frequently outpaces the model's current capabilities, leading to critically sparse reward signals and stalled learning progress. This challenge is particularly acute for ","authors_text":"Cheng Gong, Dandan Tu, Qingfu Zhang, Ran Chen, Rui Liu, Shoubo Hu, Suiyun Zhang, Xinyu Fu, Yaofang Liu, Ziru Liu","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-14T08:10:00Z","title":"GHPO: Adaptive Guidance for Stable and Efficient LLM Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.10628","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:b5a27c4a8faa1b8507d57f1b8533ac69e6ab0efc740a14100c5003592be4a9ce","target":"record","created_at":"2026-07-05T11:38:09Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"2a9a34f39ca0decba791a92f9733d4b9047bc2f6194ba1c13879dfc26979a3f2","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-14T08:10:00Z","title_canon_sha256":"fa2bdc12fbf36674e897298b2517d06c69c7195498b60e31be9ebc963c5dffa9"},"schema_version":"1.0","source":{"id":"2507.10628","kind":"arxiv","version":2}},"canonical_sha256":"0d8572bba669bc0bc8a257b1590db0369bef7b077bf494f6de4448203094df6d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"0d8572bba669bc0bc8a257b1590db0369bef7b077bf494f6de4448203094df6d","first_computed_at":"2026-07-05T11:38:09.649964Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:38:09.649964Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"uZezRo0gYe3nXf+dzK5Ci88DoR90Y1fPho3hwKqzqTEumkEvQ/oGtPktLevHx/BhB+2tB3HnCy0dXL99NXBECg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:38:09.650540Z","signed_message":"canonical_sha256_bytes"},"source_id":"2507.10628","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:b5a27c4a8faa1b8507d57f1b8533ac69e6ab0efc740a14100c5003592be4a9ce","sha256:7bfaacbbdecb3df973a034df16d6bc92f923a4ebc2a4e7ff0cc48edf0f0f9da3"],"state_sha256":"54d1c4518681cdb8e635b153c14aa964e8ecfea5311836545c0d58c5a43ab41e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"IdLbjP1I5CKWypzKNFqcbTzjeOEbeqNaRa5SNPB2ASkVwUYFu8HuyC2zJgEQuvc5M2Z0u1BzNX2zI64Ry5DODA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T08:18:56.731977Z","bundle_sha256":"bbd84b86fcc833e36d9a5d2fc535836202a5facf34d8d255ac97fc00f6cd43c1"}}