{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K6JOCPLJV3TK57N4J6V5QK4V4S","short_pith_number":"pith:K6JOCPLJ","schema_version":"1.0","canonical_sha256":"5792e13d69aee6aefdbc4fabd82b95e4ba608c76fdfb4cbad9461ee3d8860c2a","source":{"kind":"arxiv","id":"2406.03678","version":1},"attestation_state":"computed","paper":{"title":"Reflective Policy Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Junliang Xing, Renye Yan, Yaozhong Gan, Zhe Wu","submitted_at":"2024-06-06T01:46:49Z","abstract_excerpt":"On-policy reinforcement learning methods, like Trust Region Policy Optimization (TRPO) and Proximal Policy Optimization (PPO), often demand extensive data per update, leading to sample inefficiency. This paper introduces Reflective Policy Optimization (RPO), a novel on-policy extension that amalgamates past and future state-action information for policy optimization. This approach empowers the agent for introspection, allowing modifications to its actions within the current state. Theoretical analysis confirms that policy performance is monotonically improved and contracts the solution space, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.03678","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-06T01:46:49Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"5347413958a157628d28b1bbf16b07d4cf685f59da2234b75aa4b67430312afe","abstract_canon_sha256":"7fdb71408e674878179d2d16d6d3e65e3cf8c150c6ebbac97b9c9f06863e21a1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:14.381631Z","signature_b64":"fQP7aXYyIq/96M3ErwCZ3nZ9ag/7yWFgSp2NcIwKI6QFkbk0QOUrcWo+NI7oassAwaF1262th5naigXSD9tZAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5792e13d69aee6aefdbc4fabd82b95e4ba608c76fdfb4cbad9461ee3d8860c2a","last_reissued_at":"2026-07-05T08:28:14.381194Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:14.381194Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reflective Policy Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Junliang Xing, Renye Yan, Yaozhong Gan, Zhe Wu","submitted_at":"2024-06-06T01:46:49Z","abstract_excerpt":"On-policy reinforcement learning methods, like Trust Region Policy Optimization (TRPO) and Proximal Policy Optimization (PPO), often demand extensive data per update, leading to sample inefficiency. This paper introduces Reflective Policy Optimization (RPO), a novel on-policy extension that amalgamates past and future state-action information for policy optimization. This approach empowers the agent for introspection, allowing modifications to its actions within the current state. Theoretical analysis confirms that policy performance is monotonically improved and contracts the solution space, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.03678","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.03678/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.03678","created_at":"2026-07-05T08:28:14.381251+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.03678v1","created_at":"2026-07-05T08:28:14.381251+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.03678","created_at":"2026-07-05T08:28:14.381251+00:00"},{"alias_kind":"pith_short_12","alias_value":"K6JOCPLJV3TK","created_at":"2026-07-05T08:28:14.381251+00:00"},{"alias_kind":"pith_short_16","alias_value":"K6JOCPLJV3TK57N4","created_at":"2026-07-05T08:28:14.381251+00:00"},{"alias_kind":"pith_short_8","alias_value":"K6JOCPLJ","created_at":"2026-07-05T08:28:14.381251+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.06768","citing_title":"Explore or Converge? Stage-Guided Per-Step Optimization for Diffusion Models","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K6JOCPLJV3TK57N4J6V5QK4V4S","json":"https://pith.science/pith/K6JOCPLJV3TK57N4J6V5QK4V4S.json","graph_json":"https://pith.science/api/pith-number/K6JOCPLJV3TK57N4J6V5QK4V4S/graph.json","events_json":"https://pith.science/api/pith-number/K6JOCPLJV3TK57N4J6V5QK4V4S/events.json","paper":"https://pith.science/paper/K6JOCPLJ"},"agent_actions":{"view_html":"https://pith.science/pith/K6JOCPLJV3TK57N4J6V5QK4V4S","download_json":"https://pith.science/pith/K6JOCPLJV3TK57N4J6V5QK4V4S.json","view_paper":"https://pith.science/paper/K6JOCPLJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.03678&json=true","fetch_graph":"https://pith.science/api/pith-number/K6JOCPLJV3TK57N4J6V5QK4V4S/graph.json","fetch_events":"https://pith.science/api/pith-number/K6JOCPLJV3TK57N4J6V5QK4V4S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K6JOCPLJV3TK57N4J6V5QK4V4S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K6JOCPLJV3TK57N4J6V5QK4V4S/action/storage_attestation","attest_author":"https://pith.science/pith/K6JOCPLJV3TK57N4J6V5QK4V4S/action/author_attestation","sign_citation":"https://pith.science/pith/K6JOCPLJV3TK57N4J6V5QK4V4S/action/citation_signature","submit_replication":"https://pith.science/pith/K6JOCPLJV3TK57N4J6V5QK4V4S/action/replication_record"}},"created_at":"2026-07-05T08:28:14.381251+00:00","updated_at":"2026-07-05T08:28:14.381251+00:00"}