{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:QXJCMMPVEUMVKDMCRSHERIMEEG","short_pith_number":"pith:QXJCMMPV","schema_version":"1.0","canonical_sha256":"85d22631f52519550d828c8e48a18421b5c530b64968d345d0e842ce662cfd5b","source":{"kind":"arxiv","id":"1909.01150","version":3},"attestation_state":"computed","paper":{"title":"Neural Policy Gradient Methods: Global Optimality and Rates of Convergence","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Lingxiao Wang, Qi Cai, Zhaoran Wang, Zhuoran Yang","submitted_at":"2019-08-29T15:38:19Z","abstract_excerpt":"Policy gradient methods with actor-critic schemes demonstrate tremendous empirical successes, especially when the actors and critics are parameterized by neural networks. However, it remains less clear whether such \"neural\" policy gradient methods converge to globally optimal policies and whether they even converge at all. We answer both the questions affirmatively in the overparameterized regime. In detail, we prove that neural natural policy gradient converges to a globally optimal policy at a sublinear rate. Also, we show that neural vanilla policy gradient converges sublinearly to a statio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1909.01150","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-08-29T15:38:19Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"5a0393eaade6b21a7e94afea85d59993ddff479dd0b5e85dcd3f5bdaaa9f567a","abstract_canon_sha256":"493f4bbb9783eeca1e26fd4df96313f3bcf079d16f2be62f1f74e638cf2a4050"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:11:41.465514Z","signature_b64":"xzbfrvRASVdsSpKk71donbi9h3vEQkoSlpii+JQH5cL1DHvgOkXaJ+r5G+8UUeKIywksgW7v2KQqXnisbAAsAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"85d22631f52519550d828c8e48a18421b5c530b64968d345d0e842ce662cfd5b","last_reissued_at":"2026-07-05T01:11:41.464946Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:11:41.464946Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Neural Policy Gradient Methods: Global Optimality and Rates of Convergence","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Lingxiao Wang, Qi Cai, Zhaoran Wang, Zhuoran Yang","submitted_at":"2019-08-29T15:38:19Z","abstract_excerpt":"Policy gradient methods with actor-critic schemes demonstrate tremendous empirical successes, especially when the actors and critics are parameterized by neural networks. However, it remains less clear whether such \"neural\" policy gradient methods converge to globally optimal policies and whether they even converge at all. We answer both the questions affirmatively in the overparameterized regime. In detail, we prove that neural natural policy gradient converges to a globally optimal policy at a sublinear rate. Also, we show that neural vanilla policy gradient converges sublinearly to a statio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1909.01150","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1909.01150/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1909.01150","created_at":"2026-07-05T01:11:41.465008+00:00"},{"alias_kind":"arxiv_version","alias_value":"1909.01150v3","created_at":"2026-07-05T01:11:41.465008+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1909.01150","created_at":"2026-07-05T01:11:41.465008+00:00"},{"alias_kind":"pith_short_12","alias_value":"QXJCMMPVEUMV","created_at":"2026-07-05T01:11:41.465008+00:00"},{"alias_kind":"pith_short_16","alias_value":"QXJCMMPVEUMVKDMC","created_at":"2026-07-05T01:11:41.465008+00:00"},{"alias_kind":"pith_short_8","alias_value":"QXJCMMPV","created_at":"2026-07-05T01:11:41.465008+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18591","citing_title":"Randomized Advantage Transformation (RAT): Computing Natural Policy Gradients via Direct Backpropagation","ref_index":112,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01505","citing_title":"Optimal Sample Complexity for Single Time-Scale Actor-Critic with Momentum","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10909","citing_title":"Revisiting Policy Gradients for Restricted Policy Classes: Escaping Myopic Local Optima with $k$-step Policy Gradients","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09209","citing_title":"Select-then-differentiate: Solving Bilevel Optimization with Manifold Lower-level Solution Sets","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08131","citing_title":"Interactive Inverse Reinforcement Learning of Interaction Scenarios via Bi-level Optimization","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QXJCMMPVEUMVKDMCRSHERIMEEG","json":"https://pith.science/pith/QXJCMMPVEUMVKDMCRSHERIMEEG.json","graph_json":"https://pith.science/api/pith-number/QXJCMMPVEUMVKDMCRSHERIMEEG/graph.json","events_json":"https://pith.science/api/pith-number/QXJCMMPVEUMVKDMCRSHERIMEEG/events.json","paper":"https://pith.science/paper/QXJCMMPV"},"agent_actions":{"view_html":"https://pith.science/pith/QXJCMMPVEUMVKDMCRSHERIMEEG","download_json":"https://pith.science/pith/QXJCMMPVEUMVKDMCRSHERIMEEG.json","view_paper":"https://pith.science/paper/QXJCMMPV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1909.01150&json=true","fetch_graph":"https://pith.science/api/pith-number/QXJCMMPVEUMVKDMCRSHERIMEEG/graph.json","fetch_events":"https://pith.science/api/pith-number/QXJCMMPVEUMVKDMCRSHERIMEEG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QXJCMMPVEUMVKDMCRSHERIMEEG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QXJCMMPVEUMVKDMCRSHERIMEEG/action/storage_attestation","attest_author":"https://pith.science/pith/QXJCMMPVEUMVKDMCRSHERIMEEG/action/author_attestation","sign_citation":"https://pith.science/pith/QXJCMMPVEUMVKDMCRSHERIMEEG/action/citation_signature","submit_replication":"https://pith.science/pith/QXJCMMPVEUMVKDMCRSHERIMEEG/action/replication_record"}},"created_at":"2026-07-05T01:11:41.465008+00:00","updated_at":"2026-07-05T01:11:41.465008+00:00"}