{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:FN7OYUHMPSXHVKLXOVY5BKYJ56","short_pith_number":"pith:FN7OYUHM","schema_version":"1.0","canonical_sha256":"2b7eec50ec7cae7aa9777571d0ab09efa8af9585ba6ee95019fe0826fdc84eb3","source":{"kind":"arxiv","id":"2607.22982","version":1},"attestation_state":"computed","paper":{"title":"Finite-Time Analysis of the Natural Policy Gradient in Finite-Horizon Markov Decision Processes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Asha Barua, Sajad Khodadadian","submitted_at":"2026-07-25T01:37:26Z","abstract_excerpt":"Natural Policy Gradient (NPG) is a well-established Reinforcement Learning algorithm that underlies widely used methods such as Trust Region Policy Optimization and Proximal Policy Optimization, both of which have demonstrated strong empirical success. In this paper, we study exact NPG in finite-horizon Markov Decision Processes with known dynamics and horizon-dependent transition kernels. We provide the first finite-time convergence guarantees for this algorithm in this setting, for which we consider both constant and increasing step size regimes. With a constant step size $\\eta_t=\\eta$, we p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.22982","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-25T01:37:26Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"4cea04f97a688bafca5909d10cb1ae95f88300ced6ff35081c9703299f50a8f8","abstract_canon_sha256":"7811bcdb4886f559a7c6e62a613de4a52c11fbd98730f4309ecdb73c1af8f01c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-28T00:22:05.644297Z","signature_b64":"8ZZzTfYOZ3ZXf43hiSXn5POjrfRedPYHK5WQnsDJEKkSa5O5y/uBiPyQz8Cej/Lk/nSyUZPa5ULkC1P4JCXMAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b7eec50ec7cae7aa9777571d0ab09efa8af9585ba6ee95019fe0826fdc84eb3","last_reissued_at":"2026-07-28T00:22:05.642879Z","signature_status":"signed_v1","first_computed_at":"2026-07-28T00:22:05.642879Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Finite-Time Analysis of the Natural Policy Gradient in Finite-Horizon Markov Decision Processes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Asha Barua, Sajad Khodadadian","submitted_at":"2026-07-25T01:37:26Z","abstract_excerpt":"Natural Policy Gradient (NPG) is a well-established Reinforcement Learning algorithm that underlies widely used methods such as Trust Region Policy Optimization and Proximal Policy Optimization, both of which have demonstrated strong empirical success. In this paper, we study exact NPG in finite-horizon Markov Decision Processes with known dynamics and horizon-dependent transition kernels. We provide the first finite-time convergence guarantees for this algorithm in this setting, for which we consider both constant and increasing step size regimes. With a constant step size $\\eta_t=\\eta$, we p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.22982","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.22982/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.22982","created_at":"2026-07-28T00:22:05.643861+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.22982v1","created_at":"2026-07-28T00:22:05.643861+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.22982","created_at":"2026-07-28T00:22:05.643861+00:00"},{"alias_kind":"pith_short_12","alias_value":"FN7OYUHMPSXH","created_at":"2026-07-28T00:22:05.643861+00:00"},{"alias_kind":"pith_short_16","alias_value":"FN7OYUHMPSXHVKLX","created_at":"2026-07-28T00:22:05.643861+00:00"},{"alias_kind":"pith_short_8","alias_value":"FN7OYUHM","created_at":"2026-07-28T00:22:05.643861+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FN7OYUHMPSXHVKLXOVY5BKYJ56","json":"https://pith.science/pith/FN7OYUHMPSXHVKLXOVY5BKYJ56.json","graph_json":"https://pith.science/api/pith-number/FN7OYUHMPSXHVKLXOVY5BKYJ56/graph.json","events_json":"https://pith.science/api/pith-number/FN7OYUHMPSXHVKLXOVY5BKYJ56/events.json","paper":"https://pith.science/paper/FN7OYUHM"},"agent_actions":{"view_html":"https://pith.science/pith/FN7OYUHMPSXHVKLXOVY5BKYJ56","download_json":"https://pith.science/pith/FN7OYUHMPSXHVKLXOVY5BKYJ56.json","view_paper":"https://pith.science/paper/FN7OYUHM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.22982&json=true","fetch_graph":"https://pith.science/api/pith-number/FN7OYUHMPSXHVKLXOVY5BKYJ56/graph.json","fetch_events":"https://pith.science/api/pith-number/FN7OYUHMPSXHVKLXOVY5BKYJ56/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FN7OYUHMPSXHVKLXOVY5BKYJ56/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FN7OYUHMPSXHVKLXOVY5BKYJ56/action/storage_attestation","attest_author":"https://pith.science/pith/FN7OYUHMPSXHVKLXOVY5BKYJ56/action/author_attestation","sign_citation":"https://pith.science/pith/FN7OYUHMPSXHVKLXOVY5BKYJ56/action/citation_signature","submit_replication":"https://pith.science/pith/FN7OYUHMPSXHVKLXOVY5BKYJ56/action/replication_record"}},"created_at":"2026-07-28T00:22:05.643861+00:00","updated_at":"2026-07-28T00:22:05.643861+00:00"}