{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BQTBLZU4PHLDUZ2HEESTZ5HM34","short_pith_number":"pith:BQTBLZU4","schema_version":"1.0","canonical_sha256":"0c2615e69c79d63a674721253cf4ecdf2b263470f4b0c5eb70a3c689482b1928","source":{"kind":"arxiv","id":"2404.03372","version":2},"attestation_state":"computed","paper":{"title":"Elementary Analysis of Policy Gradient Methods","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"math.OC","authors_text":"Jiacai Liu, Ke Wei, Wenye Li","submitted_at":"2024-04-04T11:16:16Z","abstract_excerpt":"Projected policy gradient under the simplex parameterization, policy gradient and natural policy gradient under the softmax parameterization, are fundamental algorithms in reinforcement learning. There have been a flurry of recent activities in studying these algorithms from the theoretical aspect. Despite this, their convergence behavior is still not fully understood, even given the access to exact policy evaluations. In this paper, we focus on the discounted MDP setting and conduct a systematic study of the aforementioned policy optimization methods. Several novel results are presented, incl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.03372","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"math.OC","submitted_at":"2024-04-04T11:16:16Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"346e18a9279ca85bffe276daf2117706bdd0660303491c24ac3bdf0ed2992fd0","abstract_canon_sha256":"f9ac2d8c5ee824c8897da7e55ac44f523a2ce4c48464fe4911b04204cdb59e34"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:06:48.271340Z","signature_b64":"vXhiA5oqhAdkr1BvXcv7kiRb1rIS7yQDMQLBi50KQlsxxJKIwLE/En1WssoSmeeCPf86TRCxPxazYUw2M3KeBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c2615e69c79d63a674721253cf4ecdf2b263470f4b0c5eb70a3c689482b1928","last_reissued_at":"2026-07-05T08:06:48.270904Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:06:48.270904Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Elementary Analysis of Policy Gradient Methods","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"math.OC","authors_text":"Jiacai Liu, Ke Wei, Wenye Li","submitted_at":"2024-04-04T11:16:16Z","abstract_excerpt":"Projected policy gradient under the simplex parameterization, policy gradient and natural policy gradient under the softmax parameterization, are fundamental algorithms in reinforcement learning. There have been a flurry of recent activities in studying these algorithms from the theoretical aspect. Despite this, their convergence behavior is still not fully understood, even given the access to exact policy evaluations. In this paper, we focus on the discounted MDP setting and conduct a systematic study of the aforementioned policy optimization methods. Several novel results are presented, incl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.03372","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.03372/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.03372","created_at":"2026-07-05T08:06:48.270965+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.03372v2","created_at":"2026-07-05T08:06:48.270965+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.03372","created_at":"2026-07-05T08:06:48.270965+00:00"},{"alias_kind":"pith_short_12","alias_value":"BQTBLZU4PHLD","created_at":"2026-07-05T08:06:48.270965+00:00"},{"alias_kind":"pith_short_16","alias_value":"BQTBLZU4PHLDUZ2H","created_at":"2026-07-05T08:06:48.270965+00:00"},{"alias_kind":"pith_short_8","alias_value":"BQTBLZU4","created_at":"2026-07-05T08:06:48.270965+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30648","citing_title":"Convergence of Steepest Descent and Adam under Non-Uniform Smoothness","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2509.09838","citing_title":"Dissecting Discrete Soft Actor-Critic: Limitations and Principled Alternatives","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01505","citing_title":"Optimal Sample Complexity for Single Time-Scale Actor-Critic with Momentum","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25872","citing_title":"When Errors Can Be Beneficial: A Categorization of Imperfect Rewards for Policy Gradient","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BQTBLZU4PHLDUZ2HEESTZ5HM34","json":"https://pith.science/pith/BQTBLZU4PHLDUZ2HEESTZ5HM34.json","graph_json":"https://pith.science/api/pith-number/BQTBLZU4PHLDUZ2HEESTZ5HM34/graph.json","events_json":"https://pith.science/api/pith-number/BQTBLZU4PHLDUZ2HEESTZ5HM34/events.json","paper":"https://pith.science/paper/BQTBLZU4"},"agent_actions":{"view_html":"https://pith.science/pith/BQTBLZU4PHLDUZ2HEESTZ5HM34","download_json":"https://pith.science/pith/BQTBLZU4PHLDUZ2HEESTZ5HM34.json","view_paper":"https://pith.science/paper/BQTBLZU4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.03372&json=true","fetch_graph":"https://pith.science/api/pith-number/BQTBLZU4PHLDUZ2HEESTZ5HM34/graph.json","fetch_events":"https://pith.science/api/pith-number/BQTBLZU4PHLDUZ2HEESTZ5HM34/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BQTBLZU4PHLDUZ2HEESTZ5HM34/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BQTBLZU4PHLDUZ2HEESTZ5HM34/action/storage_attestation","attest_author":"https://pith.science/pith/BQTBLZU4PHLDUZ2HEESTZ5HM34/action/author_attestation","sign_citation":"https://pith.science/pith/BQTBLZU4PHLDUZ2HEESTZ5HM34/action/citation_signature","submit_replication":"https://pith.science/pith/BQTBLZU4PHLDUZ2HEESTZ5HM34/action/replication_record"}},"created_at":"2026-07-05T08:06:48.270965+00:00","updated_at":"2026-07-05T08:06:48.270965+00:00"}