{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:H47V7ZBN5TLNWBIDQPPPPCE55U","short_pith_number":"pith:H47V7ZBN","schema_version":"1.0","canonical_sha256":"3f3f5fe42decd6db050383def7889ded22259228c431f60373815201797d7b43","source":{"kind":"arxiv","id":"2307.02632","version":2},"attestation_state":"computed","paper":{"title":"Stability of Q-Learning Through Design and Optimism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SY","eess.SY","math.OC"],"primary_cat":"cs.LG","authors_text":"Sean Meyn","submitted_at":"2023-07-05T20:04:26Z","abstract_excerpt":"Q-learning has become an important part of the reinforcement learning toolkit since its introduction in the dissertation of Chris Watkins in the 1980s. The purpose of this paper is in part a tutorial on stochastic approximation and Q-learning, providing details regarding the INFORMS APS inaugural Applied Probability Trust Plenary Lecture, presented in Nancy France, June 2023.\n  The paper also presents new approaches to ensure stability and potentially accelerated convergence for these algorithms, and stochastic approximation in other settings. Two contributions are entirely new:\n  1. Stability"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.02632","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-07-05T20:04:26Z","cross_cats_sorted":["cs.SY","eess.SY","math.OC"],"title_canon_sha256":"185835ff4d773a60b85cddab1a717bb17a55b5742d0532fbb9481ae3c8acaf2b","abstract_canon_sha256":"a2b9ac1b7c83dec326a3f2c3c4fc2e09cc5965a261eaa257d4d284a78138f6f2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:43:04.045144Z","signature_b64":"qJ8ug1LSHc1LGBP4V5okmP1cc3dUW2LcIGX3AWZbmG0xXitT1T4hgvsOcsNn54mpDOO+Hy73KY8hnq+Y2ayzBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3f3f5fe42decd6db050383def7889ded22259228c431f60373815201797d7b43","last_reissued_at":"2026-07-05T06:43:04.044710Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:43:04.044710Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Stability of Q-Learning Through Design and Optimism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SY","eess.SY","math.OC"],"primary_cat":"cs.LG","authors_text":"Sean Meyn","submitted_at":"2023-07-05T20:04:26Z","abstract_excerpt":"Q-learning has become an important part of the reinforcement learning toolkit since its introduction in the dissertation of Chris Watkins in the 1980s. The purpose of this paper is in part a tutorial on stochastic approximation and Q-learning, providing details regarding the INFORMS APS inaugural Applied Probability Trust Plenary Lecture, presented in Nancy France, June 2023.\n  The paper also presents new approaches to ensure stability and potentially accelerated convergence for these algorithms, and stochastic approximation in other settings. Two contributions are entirely new:\n  1. Stability"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.02632","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.02632/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.02632","created_at":"2026-07-05T06:43:04.044770+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.02632v2","created_at":"2026-07-05T06:43:04.044770+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.02632","created_at":"2026-07-05T06:43:04.044770+00:00"},{"alias_kind":"pith_short_12","alias_value":"H47V7ZBN5TLN","created_at":"2026-07-05T06:43:04.044770+00:00"},{"alias_kind":"pith_short_16","alias_value":"H47V7ZBN5TLNWBID","created_at":"2026-07-05T06:43:04.044770+00:00"},{"alias_kind":"pith_short_8","alias_value":"H47V7ZBN","created_at":"2026-07-05T06:43:04.044770+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H47V7ZBN5TLNWBIDQPPPPCE55U","json":"https://pith.science/pith/H47V7ZBN5TLNWBIDQPPPPCE55U.json","graph_json":"https://pith.science/api/pith-number/H47V7ZBN5TLNWBIDQPPPPCE55U/graph.json","events_json":"https://pith.science/api/pith-number/H47V7ZBN5TLNWBIDQPPPPCE55U/events.json","paper":"https://pith.science/paper/H47V7ZBN"},"agent_actions":{"view_html":"https://pith.science/pith/H47V7ZBN5TLNWBIDQPPPPCE55U","download_json":"https://pith.science/pith/H47V7ZBN5TLNWBIDQPPPPCE55U.json","view_paper":"https://pith.science/paper/H47V7ZBN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.02632&json=true","fetch_graph":"https://pith.science/api/pith-number/H47V7ZBN5TLNWBIDQPPPPCE55U/graph.json","fetch_events":"https://pith.science/api/pith-number/H47V7ZBN5TLNWBIDQPPPPCE55U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H47V7ZBN5TLNWBIDQPPPPCE55U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H47V7ZBN5TLNWBIDQPPPPCE55U/action/storage_attestation","attest_author":"https://pith.science/pith/H47V7ZBN5TLNWBIDQPPPPCE55U/action/author_attestation","sign_citation":"https://pith.science/pith/H47V7ZBN5TLNWBIDQPPPPCE55U/action/citation_signature","submit_replication":"https://pith.science/pith/H47V7ZBN5TLNWBIDQPPPPCE55U/action/replication_record"}},"created_at":"2026-07-05T06:43:04.044770+00:00","updated_at":"2026-07-05T06:43:04.044770+00:00"}