{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:ZC4DNFOP6XDTA5TGRDZ24FIYXE","short_pith_number":"pith:ZC4DNFOP","schema_version":"1.0","canonical_sha256":"c8b83695cff5c730766688f3ae1518b9081304ea659eb908a2e5b7fe2d2bd140","source":{"kind":"arxiv","id":"2206.01315","version":2},"attestation_state":"computed","paper":{"title":"Sample-Efficient Reinforcement Learning of Partially Observable Markov Games","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GT","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chi Jin, Csaba Szepesv\\'ari, Qinghua Liu","submitted_at":"2022-06-02T21:57:47Z","abstract_excerpt":"This paper considers the challenging tasks of Multi-Agent Reinforcement Learning (MARL) under partial observability, where each agent only sees her own individual observations and actions that reveal incomplete information about the underlying state of system. This paper studies these tasks under the general model of multiplayer general-sum Partially Observable Markov Games (POMGs), which is significantly larger than the standard model of Imperfect Information Extensive-Form Games (IIEFGs). We identify a rich subclass of POMGs -- weakly revealing POMGs -- in which sample-efficient learning is "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.01315","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-06-02T21:57:47Z","cross_cats_sorted":["cs.AI","cs.GT","stat.ML"],"title_canon_sha256":"2bdf56cd2961c440d9ee68643084f5a4de872b66b06aea52858668099ca4aca2","abstract_canon_sha256":"c9bcf5fd842ddfbfb9dbb1ca5ca1cdcd1a2ec5a197afa0ee69da56fe77ab3766"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:06:59.594003Z","signature_b64":"w3+pnOo09akizdBv94UKiJKObaOPEU7J+Z/QAFm29gQuvsfVr0WTmHZ5I06+SzFO6iHx+EZ+VKGFoqyBMXfCCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c8b83695cff5c730766688f3ae1518b9081304ea659eb908a2e5b7fe2d2bd140","last_reissued_at":"2026-07-05T05:06:59.593586Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:06:59.593586Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sample-Efficient Reinforcement Learning of Partially Observable Markov Games","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GT","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chi Jin, Csaba Szepesv\\'ari, Qinghua Liu","submitted_at":"2022-06-02T21:57:47Z","abstract_excerpt":"This paper considers the challenging tasks of Multi-Agent Reinforcement Learning (MARL) under partial observability, where each agent only sees her own individual observations and actions that reveal incomplete information about the underlying state of system. This paper studies these tasks under the general model of multiplayer general-sum Partially Observable Markov Games (POMGs), which is significantly larger than the standard model of Imperfect Information Extensive-Form Games (IIEFGs). We identify a rich subclass of POMGs -- weakly revealing POMGs -- in which sample-efficient learning is "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.01315","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.01315/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.01315","created_at":"2026-07-05T05:06:59.593641+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.01315v2","created_at":"2026-07-05T05:06:59.593641+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.01315","created_at":"2026-07-05T05:06:59.593641+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZC4DNFOP6XDT","created_at":"2026-07-05T05:06:59.593641+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZC4DNFOP6XDTA5TG","created_at":"2026-07-05T05:06:59.593641+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZC4DNFOP","created_at":"2026-07-05T05:06:59.593641+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZC4DNFOP6XDTA5TGRDZ24FIYXE","json":"https://pith.science/pith/ZC4DNFOP6XDTA5TGRDZ24FIYXE.json","graph_json":"https://pith.science/api/pith-number/ZC4DNFOP6XDTA5TGRDZ24FIYXE/graph.json","events_json":"https://pith.science/api/pith-number/ZC4DNFOP6XDTA5TGRDZ24FIYXE/events.json","paper":"https://pith.science/paper/ZC4DNFOP"},"agent_actions":{"view_html":"https://pith.science/pith/ZC4DNFOP6XDTA5TGRDZ24FIYXE","download_json":"https://pith.science/pith/ZC4DNFOP6XDTA5TGRDZ24FIYXE.json","view_paper":"https://pith.science/paper/ZC4DNFOP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.01315&json=true","fetch_graph":"https://pith.science/api/pith-number/ZC4DNFOP6XDTA5TGRDZ24FIYXE/graph.json","fetch_events":"https://pith.science/api/pith-number/ZC4DNFOP6XDTA5TGRDZ24FIYXE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZC4DNFOP6XDTA5TGRDZ24FIYXE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZC4DNFOP6XDTA5TGRDZ24FIYXE/action/storage_attestation","attest_author":"https://pith.science/pith/ZC4DNFOP6XDTA5TGRDZ24FIYXE/action/author_attestation","sign_citation":"https://pith.science/pith/ZC4DNFOP6XDTA5TGRDZ24FIYXE/action/citation_signature","submit_replication":"https://pith.science/pith/ZC4DNFOP6XDTA5TGRDZ24FIYXE/action/replication_record"}},"created_at":"2026-07-05T05:06:59.593641+00:00","updated_at":"2026-07-05T05:06:59.593641+00:00"}