{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:WG5LEPJVT2KPJ2N53AOL77FIK2","short_pith_number":"pith:WG5LEPJV","schema_version":"1.0","canonical_sha256":"b1bab23d359e94f4e9bdd81cbffca856b45aee8680aaea3685e0feffaff11cd8","source":{"kind":"arxiv","id":"1906.10306","version":3},"attestation_state":"computed","paper":{"title":"Neural Proximal/Trust Region Policy Optimization Attains Globally Optimal Policy","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Boyi Liu, Qi Cai, Zhaoran Wang, Zhuoran Yang","submitted_at":"2019-06-25T03:20:04Z","abstract_excerpt":"Proximal policy optimization and trust region policy optimization (PPO and TRPO) with actor and critic parametrized by neural networks achieve significant empirical success in deep reinforcement learning. However, due to nonconvexity, the global convergence of PPO and TRPO remains less understood, which separates theory from practice. In this paper, we prove that a variant of PPO and TRPO equipped with overparametrized neural networks converges to the globally optimal policy at a sublinear rate. The key to our analysis is the global convergence of infinite-dimensional mirror descent under a no"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1906.10306","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-06-25T03:20:04Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"7cb702ef1e00460ae98c7c2bd8e863108635f3605ebd0ea51d5b23e2e13fa903","abstract_canon_sha256":"ef2f03345eff6c2b374ceca41ed3e2e1b09c9c16c08f34d56ede80b796f7892c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:46:03.899243Z","signature_b64":"wQT8cVvVMJK6CHd374N0tuFhWrDWEAYBMZ9rjVOGGy1etBPUYn1UPQ0eY1r+k2VS7pVeec8f2grTcWQdCIb9BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b1bab23d359e94f4e9bdd81cbffca856b45aee8680aaea3685e0feffaff11cd8","last_reissued_at":"2026-07-05T05:46:03.898747Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:46:03.898747Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Neural Proximal/Trust Region Policy Optimization Attains Globally Optimal Policy","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Boyi Liu, Qi Cai, Zhaoran Wang, Zhuoran Yang","submitted_at":"2019-06-25T03:20:04Z","abstract_excerpt":"Proximal policy optimization and trust region policy optimization (PPO and TRPO) with actor and critic parametrized by neural networks achieve significant empirical success in deep reinforcement learning. However, due to nonconvexity, the global convergence of PPO and TRPO remains less understood, which separates theory from practice. In this paper, we prove that a variant of PPO and TRPO equipped with overparametrized neural networks converges to the globally optimal policy at a sublinear rate. The key to our analysis is the global convergence of infinite-dimensional mirror descent under a no"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1906.10306","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1906.10306/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1906.10306","created_at":"2026-07-05T05:46:03.898806+00:00"},{"alias_kind":"arxiv_version","alias_value":"1906.10306v3","created_at":"2026-07-05T05:46:03.898806+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1906.10306","created_at":"2026-07-05T05:46:03.898806+00:00"},{"alias_kind":"pith_short_12","alias_value":"WG5LEPJVT2KP","created_at":"2026-07-05T05:46:03.898806+00:00"},{"alias_kind":"pith_short_16","alias_value":"WG5LEPJVT2KPJ2N5","created_at":"2026-07-05T05:46:03.898806+00:00"},{"alias_kind":"pith_short_8","alias_value":"WG5LEPJV","created_at":"2026-07-05T05:46:03.898806+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20999","citing_title":"Concentration of General Stochastic Approximation Under Heavy-Tailed Markovian Noise","ref_index":154,"is_internal_anchor":false},{"citing_arxiv_id":"2510.21060","citing_title":"On the Sample Complexity of Differentially Private Policy Optimization","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WG5LEPJVT2KPJ2N53AOL77FIK2","json":"https://pith.science/pith/WG5LEPJVT2KPJ2N53AOL77FIK2.json","graph_json":"https://pith.science/api/pith-number/WG5LEPJVT2KPJ2N53AOL77FIK2/graph.json","events_json":"https://pith.science/api/pith-number/WG5LEPJVT2KPJ2N53AOL77FIK2/events.json","paper":"https://pith.science/paper/WG5LEPJV"},"agent_actions":{"view_html":"https://pith.science/pith/WG5LEPJVT2KPJ2N53AOL77FIK2","download_json":"https://pith.science/pith/WG5LEPJVT2KPJ2N53AOL77FIK2.json","view_paper":"https://pith.science/paper/WG5LEPJV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1906.10306&json=true","fetch_graph":"https://pith.science/api/pith-number/WG5LEPJVT2KPJ2N53AOL77FIK2/graph.json","fetch_events":"https://pith.science/api/pith-number/WG5LEPJVT2KPJ2N53AOL77FIK2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WG5LEPJVT2KPJ2N53AOL77FIK2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WG5LEPJVT2KPJ2N53AOL77FIK2/action/storage_attestation","attest_author":"https://pith.science/pith/WG5LEPJVT2KPJ2N53AOL77FIK2/action/author_attestation","sign_citation":"https://pith.science/pith/WG5LEPJVT2KPJ2N53AOL77FIK2/action/citation_signature","submit_replication":"https://pith.science/pith/WG5LEPJVT2KPJ2N53AOL77FIK2/action/replication_record"}},"created_at":"2026-07-05T05:46:03.898806+00:00","updated_at":"2026-07-05T05:46:03.898806+00:00"}