{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:QT4HVCC5M3DCTBBTHIEB5VUO4W","short_pith_number":"pith:QT4HVCC5","canonical_record":{"source":{"id":"2607.06935","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"math.OC","submitted_at":"2026-07-08T02:57:22Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"ed05fb48eed192fd1a4b6c80b48b91933052fbcd756989bbbcc704a39d4eb9a4","abstract_canon_sha256":"736c404f32f9ebfb5a2984629410ddfda6b7850705cb0bff6733b83b7e1122ce"},"schema_version":"1.0"},"canonical_sha256":"84f87a885d66c62984333a081ed68ee5a0fd5b8108871774e1f5c36dbe800dae","source":{"kind":"arxiv","id":"2607.06935","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.06935","created_at":"2026-07-09T00:19:40Z"},{"alias_kind":"arxiv_version","alias_value":"2607.06935v1","created_at":"2026-07-09T00:19:40Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.06935","created_at":"2026-07-09T00:19:40Z"},{"alias_kind":"pith_short_12","alias_value":"QT4HVCC5M3DC","created_at":"2026-07-09T00:19:40Z"},{"alias_kind":"pith_short_16","alias_value":"QT4HVCC5M3DCTBBT","created_at":"2026-07-09T00:19:40Z"},{"alias_kind":"pith_short_8","alias_value":"QT4HVCC5","created_at":"2026-07-09T00:19:40Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:QT4HVCC5M3DCTBBTHIEB5VUO4W","target":"record","payload":{"canonical_record":{"source":{"id":"2607.06935","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"math.OC","submitted_at":"2026-07-08T02:57:22Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"ed05fb48eed192fd1a4b6c80b48b91933052fbcd756989bbbcc704a39d4eb9a4","abstract_canon_sha256":"736c404f32f9ebfb5a2984629410ddfda6b7850705cb0bff6733b83b7e1122ce"},"schema_version":"1.0"},"canonical_sha256":"84f87a885d66c62984333a081ed68ee5a0fd5b8108871774e1f5c36dbe800dae","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-09T00:19:40.592781Z","signature_b64":"lKYLJe9pOF/kMsqYuQN1qY67c4ZpT8XLRCkb+YMtOVFiQgA4Bm/K4EKnsRm4JNdHWzni25Z88NE2dUPldc4IBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"84f87a885d66c62984333a081ed68ee5a0fd5b8108871774e1f5c36dbe800dae","last_reissued_at":"2026-07-09T00:19:40.592363Z","signature_status":"signed_v1","first_computed_at":"2026-07-09T00:19:40.592363Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2607.06935","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-09T00:19:40Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"NJChIohGeaoZvLFC/JRcZKvRfh9pMQckmhp/mgaXvqjwS9XMo6VS0YRDv0jXIZ5lljYJuO9OL7DZm+/gLPiaCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T16:43:54.024883Z"},"content_sha256":"97fe9cee4f082145663acf399b7af6fa034c02da1799978c6d0c110f71b67713","schema_version":"1.0","event_id":"sha256:97fe9cee4f082145663acf399b7af6fa034c02da1799978c6d0c110f71b67713"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:QT4HVCC5M3DCTBBTHIEB5VUO4W","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Mathematical methods of reinforcement learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"math.OC","authors_text":"Alexander Gasnikov, Alexey Naumov, Artemy Rubtsov, Daniil Tiapkin, Denis Belomestny, Egor Gladin, Nikita Yudin, Yuri Sapronov","submitted_at":"2026-07-08T02:57:22Z","abstract_excerpt":"Reinforcement learning (RL) is increasingly grounded in tools from probability, optimization, and operator theory. This survey organizes the mathematical structures that underpin the design and analysis of modern algorithms in RL. We begin from Markov decision processes (MDPs) and the Bellman operators, emphasizing contraction mappings, monotonicity, and fixed-point theory that yield convergence guarantees and rates for value and policy iteration, and temporal-difference schemes. We then develop the optimization perspective: stochastic approximation and martingale methods, convex duality and t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.06935","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.06935/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-09T00:19:40Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"HLTbL/a+gj2+nRrGkprIh2t2pkta2YmGjB2ROMiQyTCdhqy7ks7LQmJQk5GTCDWYsxKisox6BI7Q2Sro3gyxAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T16:43:54.025914Z"},"content_sha256":"e50be7fe5262fb238fc4282bb33186764a55c2ff0d47924247dc68f4dc88ecf7","schema_version":"1.0","event_id":"sha256:e50be7fe5262fb238fc4282bb33186764a55c2ff0d47924247dc68f4dc88ecf7"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/QT4HVCC5M3DCTBBTHIEB5VUO4W/bundle.json","state_url":"https://pith.science/pith/QT4HVCC5M3DCTBBTHIEB5VUO4W/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/QT4HVCC5M3DCTBBTHIEB5VUO4W/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T16:43:54Z","links":{"resolver":"https://pith.science/pith/QT4HVCC5M3DCTBBTHIEB5VUO4W","bundle":"https://pith.science/pith/QT4HVCC5M3DCTBBTHIEB5VUO4W/bundle.json","state":"https://pith.science/pith/QT4HVCC5M3DCTBBTHIEB5VUO4W/state.json","well_known_bundle":"https://pith.science/.well-known/pith/QT4HVCC5M3DCTBBTHIEB5VUO4W/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:QT4HVCC5M3DCTBBTHIEB5VUO4W","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"736c404f32f9ebfb5a2984629410ddfda6b7850705cb0bff6733b83b7e1122ce","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"math.OC","submitted_at":"2026-07-08T02:57:22Z","title_canon_sha256":"ed05fb48eed192fd1a4b6c80b48b91933052fbcd756989bbbcc704a39d4eb9a4"},"schema_version":"1.0","source":{"id":"2607.06935","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.06935","created_at":"2026-07-09T00:19:40Z"},{"alias_kind":"arxiv_version","alias_value":"2607.06935v1","created_at":"2026-07-09T00:19:40Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.06935","created_at":"2026-07-09T00:19:40Z"},{"alias_kind":"pith_short_12","alias_value":"QT4HVCC5M3DC","created_at":"2026-07-09T00:19:40Z"},{"alias_kind":"pith_short_16","alias_value":"QT4HVCC5M3DCTBBT","created_at":"2026-07-09T00:19:40Z"},{"alias_kind":"pith_short_8","alias_value":"QT4HVCC5","created_at":"2026-07-09T00:19:40Z"}],"graph_snapshots":[{"event_id":"sha256:e50be7fe5262fb238fc4282bb33186764a55c2ff0d47924247dc68f4dc88ecf7","target":"graph","created_at":"2026-07-09T00:19:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.06935/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) is increasingly grounded in tools from probability, optimization, and operator theory. This survey organizes the mathematical structures that underpin the design and analysis of modern algorithms in RL. We begin from Markov decision processes (MDPs) and the Bellman operators, emphasizing contraction mappings, monotonicity, and fixed-point theory that yield convergence guarantees and rates for value and policy iteration, and temporal-difference schemes. We then develop the optimization perspective: stochastic approximation and martingale methods, convex duality and t","authors_text":"Alexander Gasnikov, Alexey Naumov, Artemy Rubtsov, Daniil Tiapkin, Denis Belomestny, Egor Gladin, Nikita Yudin, Yuri Sapronov","cross_cats":["cs.LG"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"math.OC","submitted_at":"2026-07-08T02:57:22Z","title":"Mathematical methods of reinforcement learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.06935","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:97fe9cee4f082145663acf399b7af6fa034c02da1799978c6d0c110f71b67713","target":"record","created_at":"2026-07-09T00:19:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"736c404f32f9ebfb5a2984629410ddfda6b7850705cb0bff6733b83b7e1122ce","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"math.OC","submitted_at":"2026-07-08T02:57:22Z","title_canon_sha256":"ed05fb48eed192fd1a4b6c80b48b91933052fbcd756989bbbcc704a39d4eb9a4"},"schema_version":"1.0","source":{"id":"2607.06935","kind":"arxiv","version":1}},"canonical_sha256":"84f87a885d66c62984333a081ed68ee5a0fd5b8108871774e1f5c36dbe800dae","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"84f87a885d66c62984333a081ed68ee5a0fd5b8108871774e1f5c36dbe800dae","first_computed_at":"2026-07-09T00:19:40.592363Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-09T00:19:40.592363Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"lKYLJe9pOF/kMsqYuQN1qY67c4ZpT8XLRCkb+YMtOVFiQgA4Bm/K4EKnsRm4JNdHWzni25Z88NE2dUPldc4IBg==","signature_status":"signed_v1","signed_at":"2026-07-09T00:19:40.592781Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.06935","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:97fe9cee4f082145663acf399b7af6fa034c02da1799978c6d0c110f71b67713","sha256:e50be7fe5262fb238fc4282bb33186764a55c2ff0d47924247dc68f4dc88ecf7"],"state_sha256":"50450938ae43bf5113465b460545deba3d21286a1abf0519660d084b4cf64bc3"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"sOMvBSdJV+aeZ9i3GVtFyZY3QfL0t6NcWEYI0agR471t8PWZFSw+nCXJgx4bXE80MU1hzrv8fkVkZHmAa9rsAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T16:43:54.042042Z","bundle_sha256":"22cca9d319e53d840aaebf4608aca7d1d484c453d384b9d1caa1656607033834"}}