{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:ZP2TEZNEVPX4GZIG5ROR26EGJK","short_pith_number":"pith:ZP2TEZNE","schema_version":"1.0","canonical_sha256":"cbf53265a4abefc36506ec5d1d78864a8b394e677a96cf457ff5fb6b032d93dc","source":{"kind":"arxiv","id":"2212.01623","version":1},"attestation_state":"computed","paper":{"title":"Smoothing Policy Iteration for Zero-sum Markov Games","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jingliang Duan, Shengbo Eben Li, Wenxuan Wang, Yangang Ren, Yao Lyu, Zeyang Li","submitted_at":"2022-12-03T14:39:06Z","abstract_excerpt":"Zero-sum Markov Games (MGs) has been an efficient framework for multi-agent systems and robust control, wherein a minimax problem is constructed to solve the equilibrium policies. At present, this formulation is well studied under tabular settings wherein the maximum operator is primarily and exactly solved to calculate the worst-case value function. However, it is non-trivial to extend such methods to handle complex tasks, as finding the maximum over large-scale action spaces is usually cumbersome. In this paper, we propose the smoothing policy iteration (SPI) algorithm to solve the zero-sum "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.01623","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-12-03T14:39:06Z","cross_cats_sorted":[],"title_canon_sha256":"9ce93cef0a1d69e394258cc0c32584aa434dfdae0457c9e539847cfead0b6a30","abstract_canon_sha256":"404c65c9af0255ce62b653ab1b79eddc4c214b8c076b832ccfdc05fad14f61ac"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:22:14.307772Z","signature_b64":"FBsam8FoieMrJzWjU7qCv+iglHElz34Un+EZMonXrud3acQP2s7badV1HPmU5twjkyWrAS6rNca1IwNRwyY7AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cbf53265a4abefc36506ec5d1d78864a8b394e677a96cf457ff5fb6b032d93dc","last_reissued_at":"2026-07-05T05:22:14.307335Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:22:14.307335Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Smoothing Policy Iteration for Zero-sum Markov Games","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jingliang Duan, Shengbo Eben Li, Wenxuan Wang, Yangang Ren, Yao Lyu, Zeyang Li","submitted_at":"2022-12-03T14:39:06Z","abstract_excerpt":"Zero-sum Markov Games (MGs) has been an efficient framework for multi-agent systems and robust control, wherein a minimax problem is constructed to solve the equilibrium policies. At present, this formulation is well studied under tabular settings wherein the maximum operator is primarily and exactly solved to calculate the worst-case value function. However, it is non-trivial to extend such methods to handle complex tasks, as finding the maximum over large-scale action spaces is usually cumbersome. In this paper, we propose the smoothing policy iteration (SPI) algorithm to solve the zero-sum "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.01623","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.01623/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.01623","created_at":"2026-07-05T05:22:14.307395+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.01623v1","created_at":"2026-07-05T05:22:14.307395+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.01623","created_at":"2026-07-05T05:22:14.307395+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZP2TEZNEVPX4","created_at":"2026-07-05T05:22:14.307395+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZP2TEZNEVPX4GZIG","created_at":"2026-07-05T05:22:14.307395+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZP2TEZNE","created_at":"2026-07-05T05:22:14.307395+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZP2TEZNEVPX4GZIG5ROR26EGJK","json":"https://pith.science/pith/ZP2TEZNEVPX4GZIG5ROR26EGJK.json","graph_json":"https://pith.science/api/pith-number/ZP2TEZNEVPX4GZIG5ROR26EGJK/graph.json","events_json":"https://pith.science/api/pith-number/ZP2TEZNEVPX4GZIG5ROR26EGJK/events.json","paper":"https://pith.science/paper/ZP2TEZNE"},"agent_actions":{"view_html":"https://pith.science/pith/ZP2TEZNEVPX4GZIG5ROR26EGJK","download_json":"https://pith.science/pith/ZP2TEZNEVPX4GZIG5ROR26EGJK.json","view_paper":"https://pith.science/paper/ZP2TEZNE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.01623&json=true","fetch_graph":"https://pith.science/api/pith-number/ZP2TEZNEVPX4GZIG5ROR26EGJK/graph.json","fetch_events":"https://pith.science/api/pith-number/ZP2TEZNEVPX4GZIG5ROR26EGJK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZP2TEZNEVPX4GZIG5ROR26EGJK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZP2TEZNEVPX4GZIG5ROR26EGJK/action/storage_attestation","attest_author":"https://pith.science/pith/ZP2TEZNEVPX4GZIG5ROR26EGJK/action/author_attestation","sign_citation":"https://pith.science/pith/ZP2TEZNEVPX4GZIG5ROR26EGJK/action/citation_signature","submit_replication":"https://pith.science/pith/ZP2TEZNEVPX4GZIG5ROR26EGJK/action/replication_record"}},"created_at":"2026-07-05T05:22:14.307395+00:00","updated_at":"2026-07-05T05:22:14.307395+00:00"}