{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:5ZCLTWSEVZB7USCX5MY3IXPFG6","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"8fa5d474a84062fac8b4315df1672e739237ff4a1e373cd4d83faa5e11cbc88a","cross_cats_sorted":["math.OC","stat.ML"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-05-11T23:02:46Z","title_canon_sha256":"55f742a61e8d4edb5f064e867e1e9df4e2c7cfb51dda05bd4d7ecdd6e9e0a647"},"schema_version":"1.0","source":{"id":"2205.05800","kind":"arxiv","version":6}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2205.05800","created_at":"2026-07-05T09:12:51Z"},{"alias_kind":"arxiv_version","alias_value":"2205.05800v6","created_at":"2026-07-05T09:12:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.05800","created_at":"2026-07-05T09:12:51Z"},{"alias_kind":"pith_short_12","alias_value":"5ZCLTWSEVZB7","created_at":"2026-07-05T09:12:51Z"},{"alias_kind":"pith_short_16","alias_value":"5ZCLTWSEVZB7USCX","created_at":"2026-07-05T09:12:51Z"},{"alias_kind":"pith_short_8","alias_value":"5ZCLTWSE","created_at":"2026-07-05T09:12:51Z"}],"graph_snapshots":[{"event_id":"sha256:cbc6ef40af8b93c6969857b1379b4072d172a316e024be4b353c1a6ff0dca5c2","target":"graph","created_at":"2026-07-05T09:12:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2205.05800/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We study average-reward Markov decision processes (AMDPs) and develop novel first-order methods with strong theoretical guarantees for both policy optimization and policy evaluation. Compared with intensive research efforts in finite sample analysis of policy gradient methods for discounted MDPs, existing studies on policy gradient methods for AMDPs mostly focus on regret bounds under restrictive assumptions, and they often lack guarantees on the overall sample complexities. Towards this end, we develop an average-reward stochastic policy mirror descent (SPMD) method for solving AMDPs with and","authors_text":"Feiyang Wu, Guanghui Lan, Tianjiao Li","cross_cats":["math.OC","stat.ML"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-05-11T23:02:46Z","title":"Stochastic first-order methods for average-reward Markov decision processes"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.05800","kind":"arxiv","version":6},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:17dc83277b5f27326f9b8e82848d0737e57293d1ae750bad8714d2479bfb2c91","target":"record","created_at":"2026-07-05T09:12:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"8fa5d474a84062fac8b4315df1672e739237ff4a1e373cd4d83faa5e11cbc88a","cross_cats_sorted":["math.OC","stat.ML"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-05-11T23:02:46Z","title_canon_sha256":"55f742a61e8d4edb5f064e867e1e9df4e2c7cfb51dda05bd4d7ecdd6e9e0a647"},"schema_version":"1.0","source":{"id":"2205.05800","kind":"arxiv","version":6}},"canonical_sha256":"ee44b9da44ae43fa4857eb31b45de537a5b41d8180da2661b015a184d8bf256a","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"ee44b9da44ae43fa4857eb31b45de537a5b41d8180da2661b015a184d8bf256a","first_computed_at":"2026-07-05T09:12:51.801182Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:12:51.801182Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"gb+CsO6UcYmn0CGWSw2WXEFwo7TYYhcd8WcBp7UodDLO0SAe9ozpV41TGzvExnS6fnhd3j9UAEFD0Phgxu6uDA==","signature_status":"signed_v1","signed_at":"2026-07-05T09:12:51.801588Z","signed_message":"canonical_sha256_bytes"},"source_id":"2205.05800","source_kind":"arxiv","source_version":6}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:17dc83277b5f27326f9b8e82848d0737e57293d1ae750bad8714d2479bfb2c91","sha256:cbc6ef40af8b93c6969857b1379b4072d172a316e024be4b353c1a6ff0dca5c2"],"state_sha256":"99147e7165b6c46863b39857c901ab62dcd8b4fe1850626df7254da5b8c862e9"}