{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:6NEMMZJRLVTW2MB4LHYT4TJQ7A","short_pith_number":"pith:6NEMMZJR","schema_version":"1.0","canonical_sha256":"f348c665315d676d303c59f13e4d30f8268ff6623eab0a6af303c9baa51466be","source":{"kind":"arxiv","id":"2111.06781","version":3},"attestation_state":"computed","paper":{"title":"Q-Learning for MDPs with General Spaces: Convergence and Near Optimality via Quantization under Weak Continuity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Ali Devran Kara, Naci Saldi, Serdar Y\\\"uksel","submitted_at":"2021-11-12T15:47:10Z","abstract_excerpt":"Reinforcement learning algorithms often require finiteness of state and action spaces in Markov decision processes (MDPs) (also called controlled Markov chains) and various efforts have been made in the literature towards the applicability of such algorithms for continuous state and action spaces. In this paper, we show that under very mild regularity conditions (in particular, involving only weak continuity of the transition kernel of an MDP), Q-learning for standard Borel MDPs via quantization of states and actions (called Quantized Q-Learning) converges to a limit, and furthermore this limi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.06781","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-11-12T15:47:10Z","cross_cats_sorted":["cs.SY","eess.SY"],"title_canon_sha256":"2d20aef564c9801d8345890e7e02fd9e57cb5cd049440d1c80b2d46b223367d1","abstract_canon_sha256":"103718bc031d0624e41104c46add42395040210a446ec9b3b87fd59fdb55a81b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:48:29.867022Z","signature_b64":"mQYxFtueqvs5AzwgNmXn9sXMfsOs1pnDJfmbPW4VurjkAtKMIhSDsCr8gtFuVgSRAphqVmKrBaLIBPE/vHDqAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f348c665315d676d303c59f13e4d30f8268ff6623eab0a6af303c9baa51466be","last_reissued_at":"2026-07-05T06:48:29.866532Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:48:29.866532Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Q-Learning for MDPs with General Spaces: Convergence and Near Optimality via Quantization under Weak Continuity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Ali Devran Kara, Naci Saldi, Serdar Y\\\"uksel","submitted_at":"2021-11-12T15:47:10Z","abstract_excerpt":"Reinforcement learning algorithms often require finiteness of state and action spaces in Markov decision processes (MDPs) (also called controlled Markov chains) and various efforts have been made in the literature towards the applicability of such algorithms for continuous state and action spaces. In this paper, we show that under very mild regularity conditions (in particular, involving only weak continuity of the transition kernel of an MDP), Q-learning for standard Borel MDPs via quantization of states and actions (called Quantized Q-Learning) converges to a limit, and furthermore this limi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.06781","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.06781/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.06781","created_at":"2026-07-05T06:48:29.866585+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.06781v3","created_at":"2026-07-05T06:48:29.866585+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.06781","created_at":"2026-07-05T06:48:29.866585+00:00"},{"alias_kind":"pith_short_12","alias_value":"6NEMMZJRLVTW","created_at":"2026-07-05T06:48:29.866585+00:00"},{"alias_kind":"pith_short_16","alias_value":"6NEMMZJRLVTW2MB4","created_at":"2026-07-05T06:48:29.866585+00:00"},{"alias_kind":"pith_short_8","alias_value":"6NEMMZJR","created_at":"2026-07-05T06:48:29.866585+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6NEMMZJRLVTW2MB4LHYT4TJQ7A","json":"https://pith.science/pith/6NEMMZJRLVTW2MB4LHYT4TJQ7A.json","graph_json":"https://pith.science/api/pith-number/6NEMMZJRLVTW2MB4LHYT4TJQ7A/graph.json","events_json":"https://pith.science/api/pith-number/6NEMMZJRLVTW2MB4LHYT4TJQ7A/events.json","paper":"https://pith.science/paper/6NEMMZJR"},"agent_actions":{"view_html":"https://pith.science/pith/6NEMMZJRLVTW2MB4LHYT4TJQ7A","download_json":"https://pith.science/pith/6NEMMZJRLVTW2MB4LHYT4TJQ7A.json","view_paper":"https://pith.science/paper/6NEMMZJR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.06781&json=true","fetch_graph":"https://pith.science/api/pith-number/6NEMMZJRLVTW2MB4LHYT4TJQ7A/graph.json","fetch_events":"https://pith.science/api/pith-number/6NEMMZJRLVTW2MB4LHYT4TJQ7A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6NEMMZJRLVTW2MB4LHYT4TJQ7A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6NEMMZJRLVTW2MB4LHYT4TJQ7A/action/storage_attestation","attest_author":"https://pith.science/pith/6NEMMZJRLVTW2MB4LHYT4TJQ7A/action/author_attestation","sign_citation":"https://pith.science/pith/6NEMMZJRLVTW2MB4LHYT4TJQ7A/action/citation_signature","submit_replication":"https://pith.science/pith/6NEMMZJRLVTW2MB4LHYT4TJQ7A/action/replication_record"}},"created_at":"2026-07-05T06:48:29.866585+00:00","updated_at":"2026-07-05T06:48:29.866585+00:00"}