{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:BRHNZMM2QAMJPOG5Y5IYFRZXUE","short_pith_number":"pith:BRHNZMM2","schema_version":"1.0","canonical_sha256":"0c4edcb19a801897b8ddc75182c737a1272e8eec3b5ed5e8a212082edacf14bc","source":{"kind":"arxiv","id":"2008.01062","version":3},"attestation_state":"computed","paper":{"title":"QPLEX: Duplex Dueling Multi-Agent Q-Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MA","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chongjie Zhang, Jianhao Wang, Terry Liu, Yang Yu, Zhizhou Ren","submitted_at":"2020-08-03T17:52:09Z","abstract_excerpt":"We explore value-based multi-agent reinforcement learning (MARL) in the popular paradigm of centralized training with decentralized execution (CTDE). CTDE has an important concept, Individual-Global-Max (IGM) principle, which requires the consistency between joint and local action selections to support efficient local decision-making. However, in order to achieve scalability, existing MARL methods either limit representation expressiveness of their value function classes or relax the IGM consistency, which may suffer from instability risk or may not perform well in complex domains. This paper "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2008.01062","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-08-03T17:52:09Z","cross_cats_sorted":["cs.AI","cs.MA","stat.ML"],"title_canon_sha256":"a55da67eb2b9a963194f0a3d986b15c05e0d213e8a3a0b8890878f49b7e2beb3","abstract_canon_sha256":"223223da691d38c91335df29ca14a1084c88cf251b44574cdc481f73e72b0f97"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:19:39.833958Z","signature_b64":"NZENy6d24LmJmAz6kI3RsGOW1qSH4XniCrkK897JT7GE/Dpfz/3TK0Wl/7Oz2ktuxAOHPoj3D/hBATt+hyhRAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c4edcb19a801897b8ddc75182c737a1272e8eec3b5ed5e8a212082edacf14bc","last_reissued_at":"2026-07-05T03:19:39.833592Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:19:39.833592Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"QPLEX: Duplex Dueling Multi-Agent Q-Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MA","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chongjie Zhang, Jianhao Wang, Terry Liu, Yang Yu, Zhizhou Ren","submitted_at":"2020-08-03T17:52:09Z","abstract_excerpt":"We explore value-based multi-agent reinforcement learning (MARL) in the popular paradigm of centralized training with decentralized execution (CTDE). CTDE has an important concept, Individual-Global-Max (IGM) principle, which requires the consistency between joint and local action selections to support efficient local decision-making. However, in order to achieve scalability, existing MARL methods either limit representation expressiveness of their value function classes or relax the IGM consistency, which may suffer from instability risk or may not perform well in complex domains. This paper "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2008.01062","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2008.01062/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2008.01062","created_at":"2026-07-05T03:19:39.833647+00:00"},{"alias_kind":"arxiv_version","alias_value":"2008.01062v3","created_at":"2026-07-05T03:19:39.833647+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2008.01062","created_at":"2026-07-05T03:19:39.833647+00:00"},{"alias_kind":"pith_short_12","alias_value":"BRHNZMM2QAMJ","created_at":"2026-07-05T03:19:39.833647+00:00"},{"alias_kind":"pith_short_16","alias_value":"BRHNZMM2QAMJPOG5","created_at":"2026-07-05T03:19:39.833647+00:00"},{"alias_kind":"pith_short_8","alias_value":"BRHNZMM2","created_at":"2026-07-05T03:19:39.833647+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26574","citing_title":"Revisiting Action Factorization for Complex Action Spaces","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11284","citing_title":"Phi-Actor-Critic: Steering General-Sum Games to Pareto-Efficient Correlated Equilibria","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08068","citing_title":"DICE: Entropy-Regularized Equilibrium Selection for Stable Multi-Agent LLM Coordination","ref_index":205,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04492","citing_title":"Episodic Memory Temporal Consistency for Cooperative Multi-Agent Reinforcement Learning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18024","citing_title":"Interaction-Breaking Adversarial Learning Framework for Robust Multi-Agent Reinforcement Learning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2502.03506","citing_title":"Optimistic {\\epsilon}-Greedy Exploration for Cooperative Multi-Agent Reinforcement Learning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02844","citing_title":"Wolfpack Adversarial Attack for Robust Multi-Agent Reinforcement Learning","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08391","citing_title":"SACHI: Structured Agent Coordination via Holistic Information Integration in Multi-Agent Reinforcement Learning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2509.15519","citing_title":"Fully Decentralized Cooperative Multi-Agent Reinforcement Learning is A Context Modeling Problem","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11880","citing_title":"Adaptive TD-Lambda for Cooperative Multi-agent Reinforcement Learning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08391","citing_title":"SACHI: Structured Agent Coordination via Holistic Information Integration in Multi-Agent Reinforcement Learning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17191","citing_title":"Do LLM-derived graph priors improve multi-agent coordination?","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BRHNZMM2QAMJPOG5Y5IYFRZXUE","json":"https://pith.science/pith/BRHNZMM2QAMJPOG5Y5IYFRZXUE.json","graph_json":"https://pith.science/api/pith-number/BRHNZMM2QAMJPOG5Y5IYFRZXUE/graph.json","events_json":"https://pith.science/api/pith-number/BRHNZMM2QAMJPOG5Y5IYFRZXUE/events.json","paper":"https://pith.science/paper/BRHNZMM2"},"agent_actions":{"view_html":"https://pith.science/pith/BRHNZMM2QAMJPOG5Y5IYFRZXUE","download_json":"https://pith.science/pith/BRHNZMM2QAMJPOG5Y5IYFRZXUE.json","view_paper":"https://pith.science/paper/BRHNZMM2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2008.01062&json=true","fetch_graph":"https://pith.science/api/pith-number/BRHNZMM2QAMJPOG5Y5IYFRZXUE/graph.json","fetch_events":"https://pith.science/api/pith-number/BRHNZMM2QAMJPOG5Y5IYFRZXUE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BRHNZMM2QAMJPOG5Y5IYFRZXUE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BRHNZMM2QAMJPOG5Y5IYFRZXUE/action/storage_attestation","attest_author":"https://pith.science/pith/BRHNZMM2QAMJPOG5Y5IYFRZXUE/action/author_attestation","sign_citation":"https://pith.science/pith/BRHNZMM2QAMJPOG5Y5IYFRZXUE/action/citation_signature","submit_replication":"https://pith.science/pith/BRHNZMM2QAMJPOG5Y5IYFRZXUE/action/replication_record"}},"created_at":"2026-07-05T03:19:39.833647+00:00","updated_at":"2026-07-05T03:19:39.833647+00:00"}