{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3PWDTNTRWPEKDNSRJ65F27AIBY","short_pith_number":"pith:3PWDTNTR","schema_version":"1.0","canonical_sha256":"dbec39b671b3c8a1b6514fba5d7c080e161f8ee663a5766d623084169eb5d341","source":{"kind":"arxiv","id":"2310.02671","version":2},"attestation_state":"computed","paper":{"title":"Beyond Stationarity: Convergence Analysis of Stochastic Softmax Policy Gradient Methods","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"math.OC","authors_text":"Leif D\\\"oring, Sara Klein, Simon Weissmann","submitted_at":"2023-10-04T09:21:01Z","abstract_excerpt":"Markov Decision Processes (MDPs) are a formal framework for modeling and solving sequential decision-making problems. In finite-time horizons such problems are relevant for instance for optimal stopping or specific supply chain problems, but also in the training of large language models. In contrast to infinite horizon MDPs optimal policies are not stationary, policies must be learned for every single epoch. In practice all parameters are often trained simultaneously, ignoring the inherent structure suggested by dynamic programming. This paper introduces a combination of dynamic programming an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.02671","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"math.OC","submitted_at":"2023-10-04T09:21:01Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"ac0e81da88b603aa091e8b67858d353eaf3737093f8904061e3267aaa7d030ae","abstract_canon_sha256":"13539abc9855db4f0237772af46ec89862eb0f96b0267b6c47ab66507b75296e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:15:40.101381Z","signature_b64":"fF56iGVVrAt4e16Im6Uktm+v1iZe8HLWYBTUtd12SYWDCwYpltfAR3Nu2wjDkeSlhgvqj5axM7R5ama4YuVECQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dbec39b671b3c8a1b6514fba5d7c080e161f8ee663a5766d623084169eb5d341","last_reissued_at":"2026-07-05T08:15:40.100853Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:15:40.100853Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Stationarity: Convergence Analysis of Stochastic Softmax Policy Gradient Methods","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"math.OC","authors_text":"Leif D\\\"oring, Sara Klein, Simon Weissmann","submitted_at":"2023-10-04T09:21:01Z","abstract_excerpt":"Markov Decision Processes (MDPs) are a formal framework for modeling and solving sequential decision-making problems. In finite-time horizons such problems are relevant for instance for optimal stopping or specific supply chain problems, but also in the training of large language models. In contrast to infinite horizon MDPs optimal policies are not stationary, policies must be learned for every single epoch. In practice all parameters are often trained simultaneously, ignoring the inherent structure suggested by dynamic programming. This paper introduces a combination of dynamic programming an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.02671","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.02671/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.02671","created_at":"2026-07-05T08:15:40.100916+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.02671v2","created_at":"2026-07-05T08:15:40.100916+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.02671","created_at":"2026-07-05T08:15:40.100916+00:00"},{"alias_kind":"pith_short_12","alias_value":"3PWDTNTRWPEK","created_at":"2026-07-05T08:15:40.100916+00:00"},{"alias_kind":"pith_short_16","alias_value":"3PWDTNTRWPEKDNSR","created_at":"2026-07-05T08:15:40.100916+00:00"},{"alias_kind":"pith_short_8","alias_value":"3PWDTNTR","created_at":"2026-07-05T08:15:40.100916+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.10378","citing_title":"Simultaneous Best-Response Dynamics in Random Potential Games","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3PWDTNTRWPEKDNSRJ65F27AIBY","json":"https://pith.science/pith/3PWDTNTRWPEKDNSRJ65F27AIBY.json","graph_json":"https://pith.science/api/pith-number/3PWDTNTRWPEKDNSRJ65F27AIBY/graph.json","events_json":"https://pith.science/api/pith-number/3PWDTNTRWPEKDNSRJ65F27AIBY/events.json","paper":"https://pith.science/paper/3PWDTNTR"},"agent_actions":{"view_html":"https://pith.science/pith/3PWDTNTRWPEKDNSRJ65F27AIBY","download_json":"https://pith.science/pith/3PWDTNTRWPEKDNSRJ65F27AIBY.json","view_paper":"https://pith.science/paper/3PWDTNTR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.02671&json=true","fetch_graph":"https://pith.science/api/pith-number/3PWDTNTRWPEKDNSRJ65F27AIBY/graph.json","fetch_events":"https://pith.science/api/pith-number/3PWDTNTRWPEKDNSRJ65F27AIBY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3PWDTNTRWPEKDNSRJ65F27AIBY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3PWDTNTRWPEKDNSRJ65F27AIBY/action/storage_attestation","attest_author":"https://pith.science/pith/3PWDTNTRWPEKDNSRJ65F27AIBY/action/author_attestation","sign_citation":"https://pith.science/pith/3PWDTNTRWPEKDNSRJ65F27AIBY/action/citation_signature","submit_replication":"https://pith.science/pith/3PWDTNTRWPEKDNSRJ65F27AIBY/action/replication_record"}},"created_at":"2026-07-05T08:15:40.100916+00:00","updated_at":"2026-07-05T08:15:40.100916+00:00"}