{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:SAXN4JWGC7MIWRMQ4KGYUKQGDR","short_pith_number":"pith:SAXN4JWG","schema_version":"1.0","canonical_sha256":"902ede26c617d88b4590e28d8a2a061c6fb313cb193025b8f78f1afc7587dd1f","source":{"kind":"arxiv","id":"2209.12430","version":2},"attestation_state":"computed","paper":{"title":"$O(T^{-1})$ Convergence of Optimistic-Follow-the-Regularized-Leader in Two-Player Zero-Sum Markov Games","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.GT"],"primary_cat":"cs.LG","authors_text":"Cong Ma, Yuepeng Yang","submitted_at":"2022-09-26T05:35:44Z","abstract_excerpt":"We prove that optimistic-follow-the-regularized-leader (OFTRL), together with smooth value updates, finds an $O(T^{-1})$-approximate Nash equilibrium in $T$ iterations for two-player zero-sum Markov games with full information. This improves the $\\tilde{O}(T^{-5/6})$ convergence rate recently shown in the paper Zhang et al (2022). The refined analysis hinges on two essential ingredients. First, the sum of the regrets of the two players, though not necessarily non-negative as in normal-form games, is approximately non-negative in Markov games. This property allows us to bound the second-order p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.12430","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-09-26T05:35:44Z","cross_cats_sorted":["cs.GT"],"title_canon_sha256":"8cf318351b723edc4fa06a508f35437e98db32486a5383f18ea45834fce4dcc1","abstract_canon_sha256":"fe93145a86104b5a401e4290f9cabfd1e96b569fb881afd9c9835157fe558546"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:40:08.843394Z","signature_b64":"+QhIKIgT/9imHMY6orTBRdqpYjATcAFWKTpGhFFrLyubs7O1MiDAZs9QJX+fhjuprD0MfBDKYJ6DKbUZyXhKBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"902ede26c617d88b4590e28d8a2a061c6fb313cb193025b8f78f1afc7587dd1f","last_reissued_at":"2026-07-05T05:40:08.842983Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:40:08.842983Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"$O(T^{-1})$ Convergence of Optimistic-Follow-the-Regularized-Leader in Two-Player Zero-Sum Markov Games","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.GT"],"primary_cat":"cs.LG","authors_text":"Cong Ma, Yuepeng Yang","submitted_at":"2022-09-26T05:35:44Z","abstract_excerpt":"We prove that optimistic-follow-the-regularized-leader (OFTRL), together with smooth value updates, finds an $O(T^{-1})$-approximate Nash equilibrium in $T$ iterations for two-player zero-sum Markov games with full information. This improves the $\\tilde{O}(T^{-5/6})$ convergence rate recently shown in the paper Zhang et al (2022). The refined analysis hinges on two essential ingredients. First, the sum of the regrets of the two players, though not necessarily non-negative as in normal-form games, is approximately non-negative in Markov games. This property allows us to bound the second-order p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.12430","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.12430/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.12430","created_at":"2026-07-05T05:40:08.843042+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.12430v2","created_at":"2026-07-05T05:40:08.843042+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.12430","created_at":"2026-07-05T05:40:08.843042+00:00"},{"alias_kind":"pith_short_12","alias_value":"SAXN4JWGC7MI","created_at":"2026-07-05T05:40:08.843042+00:00"},{"alias_kind":"pith_short_16","alias_value":"SAXN4JWGC7MIWRMQ","created_at":"2026-07-05T05:40:08.843042+00:00"},{"alias_kind":"pith_short_8","alias_value":"SAXN4JWG","created_at":"2026-07-05T05:40:08.843042+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.16120","citing_title":"Solving Zero-Sum Convex Markov Games","ref_index":127,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SAXN4JWGC7MIWRMQ4KGYUKQGDR","json":"https://pith.science/pith/SAXN4JWGC7MIWRMQ4KGYUKQGDR.json","graph_json":"https://pith.science/api/pith-number/SAXN4JWGC7MIWRMQ4KGYUKQGDR/graph.json","events_json":"https://pith.science/api/pith-number/SAXN4JWGC7MIWRMQ4KGYUKQGDR/events.json","paper":"https://pith.science/paper/SAXN4JWG"},"agent_actions":{"view_html":"https://pith.science/pith/SAXN4JWGC7MIWRMQ4KGYUKQGDR","download_json":"https://pith.science/pith/SAXN4JWGC7MIWRMQ4KGYUKQGDR.json","view_paper":"https://pith.science/paper/SAXN4JWG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.12430&json=true","fetch_graph":"https://pith.science/api/pith-number/SAXN4JWGC7MIWRMQ4KGYUKQGDR/graph.json","fetch_events":"https://pith.science/api/pith-number/SAXN4JWGC7MIWRMQ4KGYUKQGDR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SAXN4JWGC7MIWRMQ4KGYUKQGDR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SAXN4JWGC7MIWRMQ4KGYUKQGDR/action/storage_attestation","attest_author":"https://pith.science/pith/SAXN4JWGC7MIWRMQ4KGYUKQGDR/action/author_attestation","sign_citation":"https://pith.science/pith/SAXN4JWGC7MIWRMQ4KGYUKQGDR/action/citation_signature","submit_replication":"https://pith.science/pith/SAXN4JWGC7MIWRMQ4KGYUKQGDR/action/replication_record"}},"created_at":"2026-07-05T05:40:08.843042+00:00","updated_at":"2026-07-05T05:40:08.843042+00:00"}