{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:4ATESDGYP7VDEJVJZYDDMWTK5I","short_pith_number":"pith:4ATESDGY","canonical_record":{"source":{"id":"2210.01050","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.GT","submitted_at":"2022-10-03T16:05:43Z","cross_cats_sorted":["cs.AI","cs.LG","math.OC"],"title_canon_sha256":"e66497ee00ae3abd8eb1273dc0e2159b627c716d063bd999672cc72401775de9","abstract_canon_sha256":"d4588b15e9ebf3ca26529948984057b41496f0ea134cd4ffdcdc54dad3dd2e2e"},"schema_version":"1.0"},"canonical_sha256":"e026490cd87fea3226a9ce06365a6aea112685b333744f10718c22362c30411e","source":{"kind":"arxiv","id":"2210.01050","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2210.01050","created_at":"2026-07-05T05:03:07Z"},{"alias_kind":"arxiv_version","alias_value":"2210.01050v2","created_at":"2026-07-05T05:03:07Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.01050","created_at":"2026-07-05T05:03:07Z"},{"alias_kind":"pith_short_12","alias_value":"4ATESDGYP7VD","created_at":"2026-07-05T05:03:07Z"},{"alias_kind":"pith_short_16","alias_value":"4ATESDGYP7VDEJVJ","created_at":"2026-07-05T05:03:07Z"},{"alias_kind":"pith_short_8","alias_value":"4ATESDGY","created_at":"2026-07-05T05:03:07Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:4ATESDGYP7VDEJVJZYDDMWTK5I","target":"record","payload":{"canonical_record":{"source":{"id":"2210.01050","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.GT","submitted_at":"2022-10-03T16:05:43Z","cross_cats_sorted":["cs.AI","cs.LG","math.OC"],"title_canon_sha256":"e66497ee00ae3abd8eb1273dc0e2159b627c716d063bd999672cc72401775de9","abstract_canon_sha256":"d4588b15e9ebf3ca26529948984057b41496f0ea134cd4ffdcdc54dad3dd2e2e"},"schema_version":"1.0"},"canonical_sha256":"e026490cd87fea3226a9ce06365a6aea112685b333744f10718c22362c30411e","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:03:07.562568Z","signature_b64":"BhKsEACS3/sQrix1emrl7azyPNESgLSDoiI1KOMAqkK+9FB8tMbjjHhYvM5FhhQ6XUlgUq1a+CF7aFQlQBBtDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e026490cd87fea3226a9ce06365a6aea112685b333744f10718c22362c30411e","last_reissued_at":"2026-07-05T05:03:07.562041Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:03:07.562041Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2210.01050","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:03:07Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"95jnaCTRU9jhWPc96aPh902eJB9aXQDywiiNNJ+Ri0T+QNJyITwYOb4Kt2vNDBk03jLVWKTstxGGRsZPPt3XAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T18:53:16.447511Z"},"content_sha256":"4873c18d4527ce73b8425a3ca4dd3093187cb357d2af21a6e5a81dc0910ab409","schema_version":"1.0","event_id":"sha256:4873c18d4527ce73b8425a3ca4dd3093187cb357d2af21a6e5a81dc0910ab409"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:4ATESDGYP7VDEJVJZYDDMWTK5I","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Faster Last-iterate Convergence of Policy Optimization in Zero-Sum Markov Games","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","math.OC"],"primary_cat":"cs.GT","authors_text":"Lin Xiao, Shicong Cen, Simon S. Du, Yuejie Chi","submitted_at":"2022-10-03T16:05:43Z","abstract_excerpt":"Multi-Agent Reinforcement Learning (MARL) -- where multiple agents learn to interact in a shared dynamic environment -- permeates across a wide range of critical applications. While there has been substantial progress on understanding the global convergence of policy optimization methods in single-agent RL, designing and analysis of efficient policy optimization algorithms in the MARL setting present significant challenges, which unfortunately, remain highly inadequately addressed by existing theory. In this paper, we focus on the most basic setting of competitive multi-agent RL, namely two-pl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.01050","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.01050/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:03:07Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ukkUr2xjBogalLbvs2D5yFhoOeSqFZu+f4OUrnHG3l4MHykt/ZVUhc5Jb1SUiwzE2ee+rvV8dxzuicE4CmDOAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T18:53:16.448130Z"},"content_sha256":"6823e1ec026acd2753e1dcc540bffa3f8ebed398f9381f697d4f99fb32c42aa7","schema_version":"1.0","event_id":"sha256:6823e1ec026acd2753e1dcc540bffa3f8ebed398f9381f697d4f99fb32c42aa7"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/4ATESDGYP7VDEJVJZYDDMWTK5I/bundle.json","state_url":"https://pith.science/pith/4ATESDGYP7VDEJVJZYDDMWTK5I/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/4ATESDGYP7VDEJVJZYDDMWTK5I/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T18:53:16Z","links":{"resolver":"https://pith.science/pith/4ATESDGYP7VDEJVJZYDDMWTK5I","bundle":"https://pith.science/pith/4ATESDGYP7VDEJVJZYDDMWTK5I/bundle.json","state":"https://pith.science/pith/4ATESDGYP7VDEJVJZYDDMWTK5I/state.json","well_known_bundle":"https://pith.science/.well-known/pith/4ATESDGYP7VDEJVJZYDDMWTK5I/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:4ATESDGYP7VDEJVJZYDDMWTK5I","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d4588b15e9ebf3ca26529948984057b41496f0ea134cd4ffdcdc54dad3dd2e2e","cross_cats_sorted":["cs.AI","cs.LG","math.OC"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.GT","submitted_at":"2022-10-03T16:05:43Z","title_canon_sha256":"e66497ee00ae3abd8eb1273dc0e2159b627c716d063bd999672cc72401775de9"},"schema_version":"1.0","source":{"id":"2210.01050","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2210.01050","created_at":"2026-07-05T05:03:07Z"},{"alias_kind":"arxiv_version","alias_value":"2210.01050v2","created_at":"2026-07-05T05:03:07Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.01050","created_at":"2026-07-05T05:03:07Z"},{"alias_kind":"pith_short_12","alias_value":"4ATESDGYP7VD","created_at":"2026-07-05T05:03:07Z"},{"alias_kind":"pith_short_16","alias_value":"4ATESDGYP7VDEJVJ","created_at":"2026-07-05T05:03:07Z"},{"alias_kind":"pith_short_8","alias_value":"4ATESDGY","created_at":"2026-07-05T05:03:07Z"}],"graph_snapshots":[{"event_id":"sha256:6823e1ec026acd2753e1dcc540bffa3f8ebed398f9381f697d4f99fb32c42aa7","target":"graph","created_at":"2026-07-05T05:03:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2210.01050/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Multi-Agent Reinforcement Learning (MARL) -- where multiple agents learn to interact in a shared dynamic environment -- permeates across a wide range of critical applications. While there has been substantial progress on understanding the global convergence of policy optimization methods in single-agent RL, designing and analysis of efficient policy optimization algorithms in the MARL setting present significant challenges, which unfortunately, remain highly inadequately addressed by existing theory. In this paper, we focus on the most basic setting of competitive multi-agent RL, namely two-pl","authors_text":"Lin Xiao, Shicong Cen, Simon S. Du, Yuejie Chi","cross_cats":["cs.AI","cs.LG","math.OC"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.GT","submitted_at":"2022-10-03T16:05:43Z","title":"Faster Last-iterate Convergence of Policy Optimization in Zero-Sum Markov Games"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.01050","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4873c18d4527ce73b8425a3ca4dd3093187cb357d2af21a6e5a81dc0910ab409","target":"record","created_at":"2026-07-05T05:03:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d4588b15e9ebf3ca26529948984057b41496f0ea134cd4ffdcdc54dad3dd2e2e","cross_cats_sorted":["cs.AI","cs.LG","math.OC"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.GT","submitted_at":"2022-10-03T16:05:43Z","title_canon_sha256":"e66497ee00ae3abd8eb1273dc0e2159b627c716d063bd999672cc72401775de9"},"schema_version":"1.0","source":{"id":"2210.01050","kind":"arxiv","version":2}},"canonical_sha256":"e026490cd87fea3226a9ce06365a6aea112685b333744f10718c22362c30411e","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"e026490cd87fea3226a9ce06365a6aea112685b333744f10718c22362c30411e","first_computed_at":"2026-07-05T05:03:07.562041Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:03:07.562041Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"BhKsEACS3/sQrix1emrl7azyPNESgLSDoiI1KOMAqkK+9FB8tMbjjHhYvM5FhhQ6XUlgUq1a+CF7aFQlQBBtDg==","signature_status":"signed_v1","signed_at":"2026-07-05T05:03:07.562568Z","signed_message":"canonical_sha256_bytes"},"source_id":"2210.01050","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4873c18d4527ce73b8425a3ca4dd3093187cb357d2af21a6e5a81dc0910ab409","sha256:6823e1ec026acd2753e1dcc540bffa3f8ebed398f9381f697d4f99fb32c42aa7"],"state_sha256":"266f60b1ee70de75c9eb3abd0fa635f596ba1351ea4342dfe4291a8e0c9d8676"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"gAq/zPt0EeIXmPXBRLkA1Ag9ULYV2aNUjHaGE49zOz2w50kSJOwcA2cKePRfEov910nnHl+v6aJwQnHoqNhcBg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T18:53:16.454512Z","bundle_sha256":"5fe2f481c5a0f26c8baed25e5058cf48be6d86aec53515d516314c22b2b12356"}}