{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:A4UYMXXIALBRYBGEX2ZWWCIG5Q","short_pith_number":"pith:A4UYMXXI","schema_version":"1.0","canonical_sha256":"0729865ee802c31c04c4beb36b0906ec027fb03f7afe9c7d08344f831303667b","source":{"kind":"arxiv","id":"2212.07536","version":1},"attestation_state":"computed","paper":{"title":"Robust Policy Optimization in Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Md Masudur Rahman, Yexiang Xue","submitted_at":"2022-12-14T22:43:56Z","abstract_excerpt":"The policy gradient method enjoys the simplicity of the objective where the agent optimizes the cumulative reward directly. Moreover, in the continuous action domain, parameterized distribution of action distribution allows easy control of exploration, resulting from the variance of the representing distribution. Entropy can play an essential role in policy optimization by selecting the stochastic policy, which eventually helps better explore the environment in reinforcement learning (RL). However, the stochasticity often reduces as the training progresses; thus, the policy becomes less explor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.07536","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-12-14T22:43:56Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"be6b3e87e38255c7c9c59e261497a68349a9a186f7e17e9a991111b8e7b801bd","abstract_canon_sha256":"23d19eabda51bc25c09893121764f169c7c4fec0a7023d9d1456cee2f5e024c9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:25:33.953301Z","signature_b64":"2gejJQcGxearnWn3Ap0MjTsxjOjA9KT8I4qa2eCuGvEThy7Lp3ZyDLYmkhPiNkMqlGLQtX5lWZslKkrWoJ0XAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0729865ee802c31c04c4beb36b0906ec027fb03f7afe9c7d08344f831303667b","last_reissued_at":"2026-07-05T05:25:33.952838Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:25:33.952838Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Robust Policy Optimization in Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Md Masudur Rahman, Yexiang Xue","submitted_at":"2022-12-14T22:43:56Z","abstract_excerpt":"The policy gradient method enjoys the simplicity of the objective where the agent optimizes the cumulative reward directly. Moreover, in the continuous action domain, parameterized distribution of action distribution allows easy control of exploration, resulting from the variance of the representing distribution. Entropy can play an essential role in policy optimization by selecting the stochastic policy, which eventually helps better explore the environment in reinforcement learning (RL). However, the stochasticity often reduces as the training progresses; thus, the policy becomes less explor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.07536","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.07536/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.07536","created_at":"2026-07-05T05:25:33.952893+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.07536v1","created_at":"2026-07-05T05:25:33.952893+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.07536","created_at":"2026-07-05T05:25:33.952893+00:00"},{"alias_kind":"pith_short_12","alias_value":"A4UYMXXIALBR","created_at":"2026-07-05T05:25:33.952893+00:00"},{"alias_kind":"pith_short_16","alias_value":"A4UYMXXIALBRYBGE","created_at":"2026-07-05T05:25:33.952893+00:00"},{"alias_kind":"pith_short_8","alias_value":"A4UYMXXI","created_at":"2026-07-05T05:25:33.952893+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20644","citing_title":"Design for Manufacturing: A Manufacturability Knowledge-Integrated Reinforcement Learning Framework for Free-Form Pipe Routing in Aeroengines","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2506.07339","citing_title":"Real-Time Execution of Action Chunking Flow Policies","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A4UYMXXIALBRYBGEX2ZWWCIG5Q","json":"https://pith.science/pith/A4UYMXXIALBRYBGEX2ZWWCIG5Q.json","graph_json":"https://pith.science/api/pith-number/A4UYMXXIALBRYBGEX2ZWWCIG5Q/graph.json","events_json":"https://pith.science/api/pith-number/A4UYMXXIALBRYBGEX2ZWWCIG5Q/events.json","paper":"https://pith.science/paper/A4UYMXXI"},"agent_actions":{"view_html":"https://pith.science/pith/A4UYMXXIALBRYBGEX2ZWWCIG5Q","download_json":"https://pith.science/pith/A4UYMXXIALBRYBGEX2ZWWCIG5Q.json","view_paper":"https://pith.science/paper/A4UYMXXI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.07536&json=true","fetch_graph":"https://pith.science/api/pith-number/A4UYMXXIALBRYBGEX2ZWWCIG5Q/graph.json","fetch_events":"https://pith.science/api/pith-number/A4UYMXXIALBRYBGEX2ZWWCIG5Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A4UYMXXIALBRYBGEX2ZWWCIG5Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A4UYMXXIALBRYBGEX2ZWWCIG5Q/action/storage_attestation","attest_author":"https://pith.science/pith/A4UYMXXIALBRYBGEX2ZWWCIG5Q/action/author_attestation","sign_citation":"https://pith.science/pith/A4UYMXXIALBRYBGEX2ZWWCIG5Q/action/citation_signature","submit_replication":"https://pith.science/pith/A4UYMXXIALBRYBGEX2ZWWCIG5Q/action/replication_record"}},"created_at":"2026-07-05T05:25:33.952893+00:00","updated_at":"2026-07-05T05:25:33.952893+00:00"}