{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:423A3QSY2E6BSHBX2VL7YYUYR3","short_pith_number":"pith:423A3QSY","schema_version":"1.0","canonical_sha256":"e6b60dc258d13c191c37d557fc62988eecb518c13ac0834717f730fe10e64347","source":{"kind":"arxiv","id":"2206.07505","version":2},"attestation_state":"computed","paper":{"title":"Revisiting Some Common Practices in Cooperative Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Chao Yu, Jiaqi Yang, Wei Fu, Yi Wu, Zelai Xu","submitted_at":"2022-06-15T13:03:05Z","abstract_excerpt":"Many advances in cooperative multi-agent reinforcement learning (MARL) are based on two common design principles: value decomposition and parameter sharing. A typical MARL algorithm of this fashion decomposes a centralized Q-function into local Q-networks with parameters shared across agents. Such an algorithmic paradigm enables centralized training and decentralized execution (CTDE) and leads to efficient learning in practice. Despite all the advantages, we revisit these two principles and show that in certain scenarios, e.g., environments with a highly multi-modal reward landscape, value dec"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.07505","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2022-06-15T13:03:05Z","cross_cats_sorted":[],"title_canon_sha256":"caae389baa7f3605f577773abeb775fc4feb8ba6def0be97154f306a339de040","abstract_canon_sha256":"6a66b5e93d1c995c842f405113a579c687b48235b82f93e57084c1899e41c706"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:46:38.454134Z","signature_b64":"KLic8Di4WAPVUBbVneIrDRPVThDaZwOdkkl3EONXZj/v16RQ1QSgUPz71WJKcnik4eApmuf8YYiC5dORO+K4CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e6b60dc258d13c191c37d557fc62988eecb518c13ac0834717f730fe10e64347","last_reissued_at":"2026-07-05T04:46:38.453694Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:46:38.453694Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Revisiting Some Common Practices in Cooperative Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Chao Yu, Jiaqi Yang, Wei Fu, Yi Wu, Zelai Xu","submitted_at":"2022-06-15T13:03:05Z","abstract_excerpt":"Many advances in cooperative multi-agent reinforcement learning (MARL) are based on two common design principles: value decomposition and parameter sharing. A typical MARL algorithm of this fashion decomposes a centralized Q-function into local Q-networks with parameters shared across agents. Such an algorithmic paradigm enables centralized training and decentralized execution (CTDE) and leads to efficient learning in practice. Despite all the advantages, we revisit these two principles and show that in certain scenarios, e.g., environments with a highly multi-modal reward landscape, value dec"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.07505","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.07505/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.07505","created_at":"2026-07-05T04:46:38.453754+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.07505v2","created_at":"2026-07-05T04:46:38.453754+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.07505","created_at":"2026-07-05T04:46:38.453754+00:00"},{"alias_kind":"pith_short_12","alias_value":"423A3QSY2E6B","created_at":"2026-07-05T04:46:38.453754+00:00"},{"alias_kind":"pith_short_16","alias_value":"423A3QSY2E6BSHBX","created_at":"2026-07-05T04:46:38.453754+00:00"},{"alias_kind":"pith_short_8","alias_value":"423A3QSY","created_at":"2026-07-05T04:46:38.453754+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.12655","citing_title":"Robust Instruction Compliance in Cooperative Multi-Agent Reinforcement Learning","ref_index":65,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/423A3QSY2E6BSHBX2VL7YYUYR3","json":"https://pith.science/pith/423A3QSY2E6BSHBX2VL7YYUYR3.json","graph_json":"https://pith.science/api/pith-number/423A3QSY2E6BSHBX2VL7YYUYR3/graph.json","events_json":"https://pith.science/api/pith-number/423A3QSY2E6BSHBX2VL7YYUYR3/events.json","paper":"https://pith.science/paper/423A3QSY"},"agent_actions":{"view_html":"https://pith.science/pith/423A3QSY2E6BSHBX2VL7YYUYR3","download_json":"https://pith.science/pith/423A3QSY2E6BSHBX2VL7YYUYR3.json","view_paper":"https://pith.science/paper/423A3QSY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.07505&json=true","fetch_graph":"https://pith.science/api/pith-number/423A3QSY2E6BSHBX2VL7YYUYR3/graph.json","fetch_events":"https://pith.science/api/pith-number/423A3QSY2E6BSHBX2VL7YYUYR3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/423A3QSY2E6BSHBX2VL7YYUYR3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/423A3QSY2E6BSHBX2VL7YYUYR3/action/storage_attestation","attest_author":"https://pith.science/pith/423A3QSY2E6BSHBX2VL7YYUYR3/action/author_attestation","sign_citation":"https://pith.science/pith/423A3QSY2E6BSHBX2VL7YYUYR3/action/citation_signature","submit_replication":"https://pith.science/pith/423A3QSY2E6BSHBX2VL7YYUYR3/action/replication_record"}},"created_at":"2026-07-05T04:46:38.453754+00:00","updated_at":"2026-07-05T04:46:38.453754+00:00"}