{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:RM7YDBGQY4NWFCJKK5A2RMD52S","short_pith_number":"pith:RM7YDBGQ","schema_version":"1.0","canonical_sha256":"8b3f8184d0c71b62892a5741a8b07dd4a4b7695df21548360f16f33d98dab366","source":{"kind":"arxiv","id":"2010.03161","version":4},"attestation_state":"computed","paper":{"title":"Model-Free Non-Stationary RL: Near-Optimal Regret and Applications in Multi-Agent RL and Inventory Control","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"David Simchi-Levi, Kaiqing Zhang, Ruihao Zhu, Tamer Ba\\c{s}ar, Weichao Mao","submitted_at":"2020-10-07T04:55:56Z","abstract_excerpt":"We consider model-free reinforcement learning (RL) in non-stationary Markov decision processes. Both the reward functions and the state transition functions are allowed to vary arbitrarily over time as long as their cumulative variations do not exceed certain variation budgets. We propose Restarted Q-Learning with Upper Confidence Bounds (RestartQ-UCB), the first model-free algorithm for non-stationary RL, and show that it outperforms existing solutions in terms of dynamic regret. Specifically, RestartQ-UCB with Freedman-type bonus terms achieves a dynamic regret bound of $\\widetilde{O}(S^{\\fr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.03161","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-10-07T04:55:56Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"02e875b3a3aa36a2cd93927d8b05e2d8109b4b8facc576171b9ffa6f7d934cbd","abstract_canon_sha256":"7d01d3acf330a29f0292cd8d24c454815821b8057ad6b8837c0b9bcd432defa6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:50:00.695798Z","signature_b64":"d0+wudYJBmo/l+9lQKCGxjR2hi2reO3CWlh+03+V1Nrvgy3xxJ+Le0JkvgYF9V8FR485IPSAamJdBoLCkMvNDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b3f8184d0c71b62892a5741a8b07dd4a4b7695df21548360f16f33d98dab366","last_reissued_at":"2026-07-05T04:50:00.695124Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:50:00.695124Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model-Free Non-Stationary RL: Near-Optimal Regret and Applications in Multi-Agent RL and Inventory Control","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"David Simchi-Levi, Kaiqing Zhang, Ruihao Zhu, Tamer Ba\\c{s}ar, Weichao Mao","submitted_at":"2020-10-07T04:55:56Z","abstract_excerpt":"We consider model-free reinforcement learning (RL) in non-stationary Markov decision processes. Both the reward functions and the state transition functions are allowed to vary arbitrarily over time as long as their cumulative variations do not exceed certain variation budgets. We propose Restarted Q-Learning with Upper Confidence Bounds (RestartQ-UCB), the first model-free algorithm for non-stationary RL, and show that it outperforms existing solutions in terms of dynamic regret. Specifically, RestartQ-UCB with Freedman-type bonus terms achieves a dynamic regret bound of $\\widetilde{O}(S^{\\fr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.03161","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.03161/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.03161","created_at":"2026-07-05T04:50:00.695208+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.03161v4","created_at":"2026-07-05T04:50:00.695208+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.03161","created_at":"2026-07-05T04:50:00.695208+00:00"},{"alias_kind":"pith_short_12","alias_value":"RM7YDBGQY4NW","created_at":"2026-07-05T04:50:00.695208+00:00"},{"alias_kind":"pith_short_16","alias_value":"RM7YDBGQY4NWFCJK","created_at":"2026-07-05T04:50:00.695208+00:00"},{"alias_kind":"pith_short_8","alias_value":"RM7YDBGQ","created_at":"2026-07-05T04:50:00.695208+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29092","citing_title":"Priced Motion Through Optimal Faces: A Normal-Fan Geometry for Non-Stationary Adversarial MDPs","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16684","citing_title":"DARLING: Detection Augmented Reinforcement Learning with Non-Stationary Guarantees","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16684","citing_title":"DARLING: Detection Augmented Reinforcement Learning with Non-Stationary Guarantees","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RM7YDBGQY4NWFCJKK5A2RMD52S","json":"https://pith.science/pith/RM7YDBGQY4NWFCJKK5A2RMD52S.json","graph_json":"https://pith.science/api/pith-number/RM7YDBGQY4NWFCJKK5A2RMD52S/graph.json","events_json":"https://pith.science/api/pith-number/RM7YDBGQY4NWFCJKK5A2RMD52S/events.json","paper":"https://pith.science/paper/RM7YDBGQ"},"agent_actions":{"view_html":"https://pith.science/pith/RM7YDBGQY4NWFCJKK5A2RMD52S","download_json":"https://pith.science/pith/RM7YDBGQY4NWFCJKK5A2RMD52S.json","view_paper":"https://pith.science/paper/RM7YDBGQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.03161&json=true","fetch_graph":"https://pith.science/api/pith-number/RM7YDBGQY4NWFCJKK5A2RMD52S/graph.json","fetch_events":"https://pith.science/api/pith-number/RM7YDBGQY4NWFCJKK5A2RMD52S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RM7YDBGQY4NWFCJKK5A2RMD52S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RM7YDBGQY4NWFCJKK5A2RMD52S/action/storage_attestation","attest_author":"https://pith.science/pith/RM7YDBGQY4NWFCJKK5A2RMD52S/action/author_attestation","sign_citation":"https://pith.science/pith/RM7YDBGQY4NWFCJKK5A2RMD52S/action/citation_signature","submit_replication":"https://pith.science/pith/RM7YDBGQY4NWFCJKK5A2RMD52S/action/replication_record"}},"created_at":"2026-07-05T04:50:00.695208+00:00","updated_at":"2026-07-05T04:50:00.695208+00:00"}