{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:WZKDJNHA7UOMZWCKHB7DDGPMQZ","short_pith_number":"pith:WZKDJNHA","schema_version":"1.0","canonical_sha256":"b65434b4e0fd1cccd84a387e3199ec867f9f896d3dde6dab90ec5086d8e667d9","source":{"kind":"arxiv","id":"1911.10635","version":2},"attestation_state":"computed","paper":{"title":"Multi-Agent Reinforcement Learning: A Selective Overview of Theories and Algorithms","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MA","stat.ML"],"primary_cat":"cs.LG","authors_text":"Kaiqing Zhang, Tamer Ba\\c{s}ar, Zhuoran Yang","submitted_at":"2019-11-24T22:50:32Z","abstract_excerpt":"Recent years have witnessed significant advances in reinforcement learning (RL), which has registered great success in solving various sequential decision-making problems in machine learning. Most of the successful RL applications, e.g., the games of Go and Poker, robotics, and autonomous driving, involve the participation of more than one single agent, which naturally fall into the realm of multi-agent RL (MARL), a domain with a relatively long history, and has recently re-emerged due to advances in single-agent RL techniques. Though empirically successful, theoretical foundations for MARL ar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1911.10635","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-11-24T22:50:32Z","cross_cats_sorted":["cs.AI","cs.MA","stat.ML"],"title_canon_sha256":"1832480b9b40de1e3d323b2d0bd2ca4bf0f01dabfec7c7be8c36d737198a23a1","abstract_canon_sha256":"e7d489ae2b7ec6790dfadc01cead0cb1d786abb435f16e52990ec47a8efbdc13"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:36:06.561582Z","signature_b64":"VQ6D8FyXzJRvl5sZGhBeD0BHfpDQyzHbgjNBVR1NmJA+lJah9ClPX9q4c8HRDz1/G8P/yRt2GuwDQVGK83vNCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b65434b4e0fd1cccd84a387e3199ec867f9f896d3dde6dab90ec5086d8e667d9","last_reissued_at":"2026-07-05T02:36:06.561099Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:36:06.561099Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-Agent Reinforcement Learning: A Selective Overview of Theories and Algorithms","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MA","stat.ML"],"primary_cat":"cs.LG","authors_text":"Kaiqing Zhang, Tamer Ba\\c{s}ar, Zhuoran Yang","submitted_at":"2019-11-24T22:50:32Z","abstract_excerpt":"Recent years have witnessed significant advances in reinforcement learning (RL), which has registered great success in solving various sequential decision-making problems in machine learning. Most of the successful RL applications, e.g., the games of Go and Poker, robotics, and autonomous driving, involve the participation of more than one single agent, which naturally fall into the realm of multi-agent RL (MARL), a domain with a relatively long history, and has recently re-emerged due to advances in single-agent RL techniques. Though empirically successful, theoretical foundations for MARL ar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1911.10635","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1911.10635/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1911.10635","created_at":"2026-07-05T02:36:06.561155+00:00"},{"alias_kind":"arxiv_version","alias_value":"1911.10635v2","created_at":"2026-07-05T02:36:06.561155+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1911.10635","created_at":"2026-07-05T02:36:06.561155+00:00"},{"alias_kind":"pith_short_12","alias_value":"WZKDJNHA7UOM","created_at":"2026-07-05T02:36:06.561155+00:00"},{"alias_kind":"pith_short_16","alias_value":"WZKDJNHA7UOMZWCK","created_at":"2026-07-05T02:36:06.561155+00:00"},{"alias_kind":"pith_short_8","alias_value":"WZKDJNHA","created_at":"2026-07-05T02:36:06.561155+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19920","citing_title":"Deep-Unfolded Coordination","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20700","citing_title":"Machine-Coached Policy Revision in Adaptive Agent-Based Regulatory Simulation: A Controller-Level Contestability Layer","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14235","citing_title":"Quantum Advantage in Multi Agent Reinforcement Learning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03569","citing_title":"Dynamic Hypergame for Task Assignment in Multi-platform Mobile Crowdsensing Under Incomplete Information","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08303","citing_title":"Stability and Sensitivity Analysis for Objective Misspecifications Among Model Predictive Game Controllers","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17240","citing_title":"Safe and Policy-Compliant Multi-Agent Orchestration for Enterprise AI","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WZKDJNHA7UOMZWCKHB7DDGPMQZ","json":"https://pith.science/pith/WZKDJNHA7UOMZWCKHB7DDGPMQZ.json","graph_json":"https://pith.science/api/pith-number/WZKDJNHA7UOMZWCKHB7DDGPMQZ/graph.json","events_json":"https://pith.science/api/pith-number/WZKDJNHA7UOMZWCKHB7DDGPMQZ/events.json","paper":"https://pith.science/paper/WZKDJNHA"},"agent_actions":{"view_html":"https://pith.science/pith/WZKDJNHA7UOMZWCKHB7DDGPMQZ","download_json":"https://pith.science/pith/WZKDJNHA7UOMZWCKHB7DDGPMQZ.json","view_paper":"https://pith.science/paper/WZKDJNHA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1911.10635&json=true","fetch_graph":"https://pith.science/api/pith-number/WZKDJNHA7UOMZWCKHB7DDGPMQZ/graph.json","fetch_events":"https://pith.science/api/pith-number/WZKDJNHA7UOMZWCKHB7DDGPMQZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WZKDJNHA7UOMZWCKHB7DDGPMQZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WZKDJNHA7UOMZWCKHB7DDGPMQZ/action/storage_attestation","attest_author":"https://pith.science/pith/WZKDJNHA7UOMZWCKHB7DDGPMQZ/action/author_attestation","sign_citation":"https://pith.science/pith/WZKDJNHA7UOMZWCKHB7DDGPMQZ/action/citation_signature","submit_replication":"https://pith.science/pith/WZKDJNHA7UOMZWCKHB7DDGPMQZ/action/replication_record"}},"created_at":"2026-07-05T02:36:06.561155+00:00","updated_at":"2026-07-05T02:36:06.561155+00:00"}