{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:CYGDKISX6MNTLHL4Q5RPRWXX77","short_pith_number":"pith:CYGDKISX","canonical_record":{"source":{"id":"2505.20686","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-27T03:58:50Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1fb199337f1363a7f88b06ff3513e0864f28fa4b6698d9d1102378d7c56b98ec","abstract_canon_sha256":"1d0abb37777415ad07bd0fc2b18ca461b3e7681f974add75539be7b4e4679797"},"schema_version":"1.0"},"canonical_sha256":"160c352257f31b359d7c8762f8daf7ffcb18e66ce857189d7b3a0d72c3878e91","source":{"kind":"arxiv","id":"2505.20686","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.20686","created_at":"2026-07-05T11:10:16Z"},{"alias_kind":"arxiv_version","alias_value":"2505.20686v1","created_at":"2026-07-05T11:10:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.20686","created_at":"2026-07-05T11:10:16Z"},{"alias_kind":"pith_short_12","alias_value":"CYGDKISX6MNT","created_at":"2026-07-05T11:10:16Z"},{"alias_kind":"pith_short_16","alias_value":"CYGDKISX6MNTLHL4","created_at":"2026-07-05T11:10:16Z"},{"alias_kind":"pith_short_8","alias_value":"CYGDKISX","created_at":"2026-07-05T11:10:16Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:CYGDKISX6MNTLHL4Q5RPRWXX77","target":"record","payload":{"canonical_record":{"source":{"id":"2505.20686","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-27T03:58:50Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1fb199337f1363a7f88b06ff3513e0864f28fa4b6698d9d1102378d7c56b98ec","abstract_canon_sha256":"1d0abb37777415ad07bd0fc2b18ca461b3e7681f974add75539be7b4e4679797"},"schema_version":"1.0"},"canonical_sha256":"160c352257f31b359d7c8762f8daf7ffcb18e66ce857189d7b3a0d72c3878e91","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:16.684528Z","signature_b64":"vLYCTNwOZNPCRguYWxRfal9CW3ZVPaYAtS5w4xZ5ohogZ9+Uk0V2iaBqYhSYgnABx6HGBtvEfbuFC+2xNmNFDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"160c352257f31b359d7c8762f8daf7ffcb18e66ce857189d7b3a0d72c3878e91","last_reissued_at":"2026-07-05T11:10:16.684070Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:16.684070Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2505.20686","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:10:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Qq0qyRYnzQQPsFALSxq7omW+5kSXHojAs47jfJZkVvloF1MZVh1svncAY9wPQlb4HLS2yCQQTAvGGyaEthXMAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T23:08:15.397974Z"},"content_sha256":"2f89eba077b2c70e8b1b30589d9762ea44079b16501ab7784dddbbbfa177653d","schema_version":"1.0","event_id":"sha256:2f89eba077b2c70e8b1b30589d9762ea44079b16501ab7784dddbbbfa177653d"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:CYGDKISX6MNTLHL4Q5RPRWXX77","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Accelerating RL for LLM Reasoning with Optimal Advantage Regression","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jason D. Lee, Kiant\\'e Brantley, Mingyu Chen, Wenhao Zhan, Wen Sun, Xuezhou Zhang, Zhaolin Gao","submitted_at":"2025-05-27T03:58:50Z","abstract_excerpt":"Reinforcement learning (RL) has emerged as a powerful tool for fine-tuning large language models (LLMs) to improve complex reasoning abilities. However, state-of-the-art policy optimization methods often suffer from high computational overhead and memory consumption, primarily due to the need for multiple generations per prompt and the reliance on critic networks or advantage estimates of the current policy. In this paper, we propose $A$*-PO, a novel two-stage policy optimization framework that directly approximates the optimal advantage function and enables efficient training of LLMs for reas"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.20686","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.20686/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:10:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"14hKOF1O7Zw0zVVAIrrGrtmm0fUKyXk57+hz55xsk37sH44oF4QbvPk4t58R3p8mDlBx8GhgwVa02iwy95eDBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T23:08:15.399029Z"},"content_sha256":"4908c2e7327cd5de5be27957f6454c6b4b8cff444c19c3989460da14093baada","schema_version":"1.0","event_id":"sha256:4908c2e7327cd5de5be27957f6454c6b4b8cff444c19c3989460da14093baada"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/CYGDKISX6MNTLHL4Q5RPRWXX77/bundle.json","state_url":"https://pith.science/pith/CYGDKISX6MNTLHL4Q5RPRWXX77/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/CYGDKISX6MNTLHL4Q5RPRWXX77/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-10T23:08:15Z","links":{"resolver":"https://pith.science/pith/CYGDKISX6MNTLHL4Q5RPRWXX77","bundle":"https://pith.science/pith/CYGDKISX6MNTLHL4Q5RPRWXX77/bundle.json","state":"https://pith.science/pith/CYGDKISX6MNTLHL4Q5RPRWXX77/state.json","well_known_bundle":"https://pith.science/.well-known/pith/CYGDKISX6MNTLHL4Q5RPRWXX77/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:CYGDKISX6MNTLHL4Q5RPRWXX77","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"1d0abb37777415ad07bd0fc2b18ca461b3e7681f974add75539be7b4e4679797","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-27T03:58:50Z","title_canon_sha256":"1fb199337f1363a7f88b06ff3513e0864f28fa4b6698d9d1102378d7c56b98ec"},"schema_version":"1.0","source":{"id":"2505.20686","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.20686","created_at":"2026-07-05T11:10:16Z"},{"alias_kind":"arxiv_version","alias_value":"2505.20686v1","created_at":"2026-07-05T11:10:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.20686","created_at":"2026-07-05T11:10:16Z"},{"alias_kind":"pith_short_12","alias_value":"CYGDKISX6MNT","created_at":"2026-07-05T11:10:16Z"},{"alias_kind":"pith_short_16","alias_value":"CYGDKISX6MNTLHL4","created_at":"2026-07-05T11:10:16Z"},{"alias_kind":"pith_short_8","alias_value":"CYGDKISX","created_at":"2026-07-05T11:10:16Z"}],"graph_snapshots":[{"event_id":"sha256:4908c2e7327cd5de5be27957f6454c6b4b8cff444c19c3989460da14093baada","target":"graph","created_at":"2026-07-05T11:10:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.20686/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) has emerged as a powerful tool for fine-tuning large language models (LLMs) to improve complex reasoning abilities. However, state-of-the-art policy optimization methods often suffer from high computational overhead and memory consumption, primarily due to the need for multiple generations per prompt and the reliance on critic networks or advantage estimates of the current policy. In this paper, we propose $A$*-PO, a novel two-stage policy optimization framework that directly approximates the optimal advantage function and enables efficient training of LLMs for reas","authors_text":"Jason D. Lee, Kiant\\'e Brantley, Mingyu Chen, Wenhao Zhan, Wen Sun, Xuezhou Zhang, Zhaolin Gao","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-27T03:58:50Z","title":"Accelerating RL for LLM Reasoning with Optimal Advantage Regression"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.20686","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:2f89eba077b2c70e8b1b30589d9762ea44079b16501ab7784dddbbbfa177653d","target":"record","created_at":"2026-07-05T11:10:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"1d0abb37777415ad07bd0fc2b18ca461b3e7681f974add75539be7b4e4679797","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-27T03:58:50Z","title_canon_sha256":"1fb199337f1363a7f88b06ff3513e0864f28fa4b6698d9d1102378d7c56b98ec"},"schema_version":"1.0","source":{"id":"2505.20686","kind":"arxiv","version":1}},"canonical_sha256":"160c352257f31b359d7c8762f8daf7ffcb18e66ce857189d7b3a0d72c3878e91","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"160c352257f31b359d7c8762f8daf7ffcb18e66ce857189d7b3a0d72c3878e91","first_computed_at":"2026-07-05T11:10:16.684070Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:10:16.684070Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"vLYCTNwOZNPCRguYWxRfal9CW3ZVPaYAtS5w4xZ5ohogZ9+Uk0V2iaBqYhSYgnABx6HGBtvEfbuFC+2xNmNFDg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:10:16.684528Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.20686","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:2f89eba077b2c70e8b1b30589d9762ea44079b16501ab7784dddbbbfa177653d","sha256:4908c2e7327cd5de5be27957f6454c6b4b8cff444c19c3989460da14093baada"],"state_sha256":"e4bec81e74a44108df59d67dd0fd4b430a0c479b57685fa307291dca3a380a90"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"SA8HydcoKvoMuAYU/MdqBi3qblYZHMP/jWNXOePWTAtn7SBZ2GtpHkvOwl1maHgF6FPAtlVtttPvepf6w/BMDg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-10T23:08:15.404514Z","bundle_sha256":"b3da196e0d1cb7bf947cd1b069516074fb9d4320a7556eef6764676a2335e788"}}