{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:5ZCLTWSEVZB7USCX5MY3IXPFG6","short_pith_number":"pith:5ZCLTWSE","canonical_record":{"source":{"id":"2205.05800","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-05-11T23:02:46Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"55f742a61e8d4edb5f064e867e1e9df4e2c7cfb51dda05bd4d7ecdd6e9e0a647","abstract_canon_sha256":"8fa5d474a84062fac8b4315df1672e739237ff4a1e373cd4d83faa5e11cbc88a"},"schema_version":"1.0"},"canonical_sha256":"ee44b9da44ae43fa4857eb31b45de537a5b41d8180da2661b015a184d8bf256a","source":{"kind":"arxiv","id":"2205.05800","version":6},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2205.05800","created_at":"2026-07-05T09:12:51Z"},{"alias_kind":"arxiv_version","alias_value":"2205.05800v6","created_at":"2026-07-05T09:12:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.05800","created_at":"2026-07-05T09:12:51Z"},{"alias_kind":"pith_short_12","alias_value":"5ZCLTWSEVZB7","created_at":"2026-07-05T09:12:51Z"},{"alias_kind":"pith_short_16","alias_value":"5ZCLTWSEVZB7USCX","created_at":"2026-07-05T09:12:51Z"},{"alias_kind":"pith_short_8","alias_value":"5ZCLTWSE","created_at":"2026-07-05T09:12:51Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:5ZCLTWSEVZB7USCX5MY3IXPFG6","target":"record","payload":{"canonical_record":{"source":{"id":"2205.05800","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-05-11T23:02:46Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"55f742a61e8d4edb5f064e867e1e9df4e2c7cfb51dda05bd4d7ecdd6e9e0a647","abstract_canon_sha256":"8fa5d474a84062fac8b4315df1672e739237ff4a1e373cd4d83faa5e11cbc88a"},"schema_version":"1.0"},"canonical_sha256":"ee44b9da44ae43fa4857eb31b45de537a5b41d8180da2661b015a184d8bf256a","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:12:51.801588Z","signature_b64":"gb+CsO6UcYmn0CGWSw2WXEFwo7TYYhcd8WcBp7UodDLO0SAe9ozpV41TGzvExnS6fnhd3j9UAEFD0Phgxu6uDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ee44b9da44ae43fa4857eb31b45de537a5b41d8180da2661b015a184d8bf256a","last_reissued_at":"2026-07-05T09:12:51.801182Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:12:51.801182Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2205.05800","source_version":6,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:12:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"sDdRrFcNcy9QIQ86sXX+bI23Bq+Z7iP3gQxKXh89g77VdvcvZYIEN8vwEeXoSRIydH3Hdn+2nN6nn6wZIHUsBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-16T10:12:02.855981Z"},"content_sha256":"17dc83277b5f27326f9b8e82848d0737e57293d1ae750bad8714d2479bfb2c91","schema_version":"1.0","event_id":"sha256:17dc83277b5f27326f9b8e82848d0737e57293d1ae750bad8714d2479bfb2c91"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:5ZCLTWSEVZB7USCX5MY3IXPFG6","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Stochastic first-order methods for average-reward Markov decision processes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Feiyang Wu, Guanghui Lan, Tianjiao Li","submitted_at":"2022-05-11T23:02:46Z","abstract_excerpt":"We study average-reward Markov decision processes (AMDPs) and develop novel first-order methods with strong theoretical guarantees for both policy optimization and policy evaluation. Compared with intensive research efforts in finite sample analysis of policy gradient methods for discounted MDPs, existing studies on policy gradient methods for AMDPs mostly focus on regret bounds under restrictive assumptions, and they often lack guarantees on the overall sample complexities. Towards this end, we develop an average-reward stochastic policy mirror descent (SPMD) method for solving AMDPs with and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.05800","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.05800/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:12:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"MaO/x7cmJ8TyqM9au8NIpEkmKmmhsOmAY4dvsT6J3rZHOBs+ZmLDGSjC9gTQqnZxhcbbgC7DDPpnQRQYTPSfDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-16T10:12:02.856612Z"},"content_sha256":"cbc6ef40af8b93c6969857b1379b4072d172a316e024be4b353c1a6ff0dca5c2","schema_version":"1.0","event_id":"sha256:cbc6ef40af8b93c6969857b1379b4072d172a316e024be4b353c1a6ff0dca5c2"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/5ZCLTWSEVZB7USCX5MY3IXPFG6/bundle.json","state_url":"https://pith.science/pith/5ZCLTWSEVZB7USCX5MY3IXPFG6/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/5ZCLTWSEVZB7USCX5MY3IXPFG6/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-16T10:12:02Z","links":{"resolver":"https://pith.science/pith/5ZCLTWSEVZB7USCX5MY3IXPFG6","bundle":"https://pith.science/pith/5ZCLTWSEVZB7USCX5MY3IXPFG6/bundle.json","state":"https://pith.science/pith/5ZCLTWSEVZB7USCX5MY3IXPFG6/state.json","well_known_bundle":"https://pith.science/.well-known/pith/5ZCLTWSEVZB7USCX5MY3IXPFG6/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:5ZCLTWSEVZB7USCX5MY3IXPFG6","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"8fa5d474a84062fac8b4315df1672e739237ff4a1e373cd4d83faa5e11cbc88a","cross_cats_sorted":["math.OC","stat.ML"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-05-11T23:02:46Z","title_canon_sha256":"55f742a61e8d4edb5f064e867e1e9df4e2c7cfb51dda05bd4d7ecdd6e9e0a647"},"schema_version":"1.0","source":{"id":"2205.05800","kind":"arxiv","version":6}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2205.05800","created_at":"2026-07-05T09:12:51Z"},{"alias_kind":"arxiv_version","alias_value":"2205.05800v6","created_at":"2026-07-05T09:12:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.05800","created_at":"2026-07-05T09:12:51Z"},{"alias_kind":"pith_short_12","alias_value":"5ZCLTWSEVZB7","created_at":"2026-07-05T09:12:51Z"},{"alias_kind":"pith_short_16","alias_value":"5ZCLTWSEVZB7USCX","created_at":"2026-07-05T09:12:51Z"},{"alias_kind":"pith_short_8","alias_value":"5ZCLTWSE","created_at":"2026-07-05T09:12:51Z"}],"graph_snapshots":[{"event_id":"sha256:cbc6ef40af8b93c6969857b1379b4072d172a316e024be4b353c1a6ff0dca5c2","target":"graph","created_at":"2026-07-05T09:12:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2205.05800/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We study average-reward Markov decision processes (AMDPs) and develop novel first-order methods with strong theoretical guarantees for both policy optimization and policy evaluation. Compared with intensive research efforts in finite sample analysis of policy gradient methods for discounted MDPs, existing studies on policy gradient methods for AMDPs mostly focus on regret bounds under restrictive assumptions, and they often lack guarantees on the overall sample complexities. Towards this end, we develop an average-reward stochastic policy mirror descent (SPMD) method for solving AMDPs with and","authors_text":"Feiyang Wu, Guanghui Lan, Tianjiao Li","cross_cats":["math.OC","stat.ML"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-05-11T23:02:46Z","title":"Stochastic first-order methods for average-reward Markov decision processes"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.05800","kind":"arxiv","version":6},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:17dc83277b5f27326f9b8e82848d0737e57293d1ae750bad8714d2479bfb2c91","target":"record","created_at":"2026-07-05T09:12:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"8fa5d474a84062fac8b4315df1672e739237ff4a1e373cd4d83faa5e11cbc88a","cross_cats_sorted":["math.OC","stat.ML"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-05-11T23:02:46Z","title_canon_sha256":"55f742a61e8d4edb5f064e867e1e9df4e2c7cfb51dda05bd4d7ecdd6e9e0a647"},"schema_version":"1.0","source":{"id":"2205.05800","kind":"arxiv","version":6}},"canonical_sha256":"ee44b9da44ae43fa4857eb31b45de537a5b41d8180da2661b015a184d8bf256a","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"ee44b9da44ae43fa4857eb31b45de537a5b41d8180da2661b015a184d8bf256a","first_computed_at":"2026-07-05T09:12:51.801182Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:12:51.801182Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"gb+CsO6UcYmn0CGWSw2WXEFwo7TYYhcd8WcBp7UodDLO0SAe9ozpV41TGzvExnS6fnhd3j9UAEFD0Phgxu6uDA==","signature_status":"signed_v1","signed_at":"2026-07-05T09:12:51.801588Z","signed_message":"canonical_sha256_bytes"},"source_id":"2205.05800","source_kind":"arxiv","source_version":6}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:17dc83277b5f27326f9b8e82848d0737e57293d1ae750bad8714d2479bfb2c91","sha256:cbc6ef40af8b93c6969857b1379b4072d172a316e024be4b353c1a6ff0dca5c2"],"state_sha256":"99147e7165b6c46863b39857c901ab62dcd8b4fe1850626df7254da5b8c862e9"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"AMjwrsDjEHvbjfFQmKt8oMrLDZaf/gyqrlHHoK0ZDCEXM6LQK5j0pfCE4UYgUFCYYozSc1vlOpA5C9Vzzx/IAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-16T10:12:02.861120Z","bundle_sha256":"42dccc618c0776aff827050cdb1c9672ae819ca660bbc3e8bf5f9feda281f3d7"}}