{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:423A3QSY2E6BSHBX2VL7YYUYR3","short_pith_number":"pith:423A3QSY","canonical_record":{"source":{"id":"2206.07505","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2022-06-15T13:03:05Z","cross_cats_sorted":[],"title_canon_sha256":"caae389baa7f3605f577773abeb775fc4feb8ba6def0be97154f306a339de040","abstract_canon_sha256":"6a66b5e93d1c995c842f405113a579c687b48235b82f93e57084c1899e41c706"},"schema_version":"1.0"},"canonical_sha256":"e6b60dc258d13c191c37d557fc62988eecb518c13ac0834717f730fe10e64347","source":{"kind":"arxiv","id":"2206.07505","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2206.07505","created_at":"2026-07-05T04:46:38Z"},{"alias_kind":"arxiv_version","alias_value":"2206.07505v2","created_at":"2026-07-05T04:46:38Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.07505","created_at":"2026-07-05T04:46:38Z"},{"alias_kind":"pith_short_12","alias_value":"423A3QSY2E6B","created_at":"2026-07-05T04:46:38Z"},{"alias_kind":"pith_short_16","alias_value":"423A3QSY2E6BSHBX","created_at":"2026-07-05T04:46:38Z"},{"alias_kind":"pith_short_8","alias_value":"423A3QSY","created_at":"2026-07-05T04:46:38Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:423A3QSY2E6BSHBX2VL7YYUYR3","target":"record","payload":{"canonical_record":{"source":{"id":"2206.07505","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2022-06-15T13:03:05Z","cross_cats_sorted":[],"title_canon_sha256":"caae389baa7f3605f577773abeb775fc4feb8ba6def0be97154f306a339de040","abstract_canon_sha256":"6a66b5e93d1c995c842f405113a579c687b48235b82f93e57084c1899e41c706"},"schema_version":"1.0"},"canonical_sha256":"e6b60dc258d13c191c37d557fc62988eecb518c13ac0834717f730fe10e64347","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:46:38.454134Z","signature_b64":"KLic8Di4WAPVUBbVneIrDRPVThDaZwOdkkl3EONXZj/v16RQ1QSgUPz71WJKcnik4eApmuf8YYiC5dORO+K4CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e6b60dc258d13c191c37d557fc62988eecb518c13ac0834717f730fe10e64347","last_reissued_at":"2026-07-05T04:46:38.453694Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:46:38.453694Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2206.07505","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T04:46:38Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"txqZCX5FCsTwEAMwpm1Qd/SBoXsxdlJj3WsQi+VOuABVohbLzndusPyhet5WKLLOBThYG7GrcjYeB7woorPYAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T12:32:39.277876Z"},"content_sha256":"0cdc16bfb5688c96ce514a74921e27212dca0b10319e58ac0ba478e52322b1d8","schema_version":"1.0","event_id":"sha256:0cdc16bfb5688c96ce514a74921e27212dca0b10319e58ac0ba478e52322b1d8"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:423A3QSY2E6BSHBX2VL7YYUYR3","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Revisiting Some Common Practices in Cooperative Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Chao Yu, Jiaqi Yang, Wei Fu, Yi Wu, Zelai Xu","submitted_at":"2022-06-15T13:03:05Z","abstract_excerpt":"Many advances in cooperative multi-agent reinforcement learning (MARL) are based on two common design principles: value decomposition and parameter sharing. A typical MARL algorithm of this fashion decomposes a centralized Q-function into local Q-networks with parameters shared across agents. Such an algorithmic paradigm enables centralized training and decentralized execution (CTDE) and leads to efficient learning in practice. Despite all the advantages, we revisit these two principles and show that in certain scenarios, e.g., environments with a highly multi-modal reward landscape, value dec"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.07505","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.07505/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T04:46:38Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"iIjxkOfqkOjYsELvpM5xqWPV7D6qmTbnO7X2v+lxRJvNdALUGWgmJBcQFPciSDa7/ceEO+W3PvgQ9lTbUg1/Cw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T12:32:39.278931Z"},"content_sha256":"b100037a6624c0c67af0adf686bbaa82e0921e8d7e3613f18a03c14cc2d0e4f7","schema_version":"1.0","event_id":"sha256:b100037a6624c0c67af0adf686bbaa82e0921e8d7e3613f18a03c14cc2d0e4f7"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/423A3QSY2E6BSHBX2VL7YYUYR3/bundle.json","state_url":"https://pith.science/pith/423A3QSY2E6BSHBX2VL7YYUYR3/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/423A3QSY2E6BSHBX2VL7YYUYR3/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-10T12:32:39Z","links":{"resolver":"https://pith.science/pith/423A3QSY2E6BSHBX2VL7YYUYR3","bundle":"https://pith.science/pith/423A3QSY2E6BSHBX2VL7YYUYR3/bundle.json","state":"https://pith.science/pith/423A3QSY2E6BSHBX2VL7YYUYR3/state.json","well_known_bundle":"https://pith.science/.well-known/pith/423A3QSY2E6BSHBX2VL7YYUYR3/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:423A3QSY2E6BSHBX2VL7YYUYR3","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"6a66b5e93d1c995c842f405113a579c687b48235b82f93e57084c1899e41c706","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2022-06-15T13:03:05Z","title_canon_sha256":"caae389baa7f3605f577773abeb775fc4feb8ba6def0be97154f306a339de040"},"schema_version":"1.0","source":{"id":"2206.07505","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2206.07505","created_at":"2026-07-05T04:46:38Z"},{"alias_kind":"arxiv_version","alias_value":"2206.07505v2","created_at":"2026-07-05T04:46:38Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.07505","created_at":"2026-07-05T04:46:38Z"},{"alias_kind":"pith_short_12","alias_value":"423A3QSY2E6B","created_at":"2026-07-05T04:46:38Z"},{"alias_kind":"pith_short_16","alias_value":"423A3QSY2E6BSHBX","created_at":"2026-07-05T04:46:38Z"},{"alias_kind":"pith_short_8","alias_value":"423A3QSY","created_at":"2026-07-05T04:46:38Z"}],"graph_snapshots":[{"event_id":"sha256:b100037a6624c0c67af0adf686bbaa82e0921e8d7e3613f18a03c14cc2d0e4f7","target":"graph","created_at":"2026-07-05T04:46:38Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2206.07505/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Many advances in cooperative multi-agent reinforcement learning (MARL) are based on two common design principles: value decomposition and parameter sharing. A typical MARL algorithm of this fashion decomposes a centralized Q-function into local Q-networks with parameters shared across agents. Such an algorithmic paradigm enables centralized training and decentralized execution (CTDE) and leads to efficient learning in practice. Despite all the advantages, we revisit these two principles and show that in certain scenarios, e.g., environments with a highly multi-modal reward landscape, value dec","authors_text":"Chao Yu, Jiaqi Yang, Wei Fu, Yi Wu, Zelai Xu","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2022-06-15T13:03:05Z","title":"Revisiting Some Common Practices in Cooperative Multi-Agent Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.07505","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:0cdc16bfb5688c96ce514a74921e27212dca0b10319e58ac0ba478e52322b1d8","target":"record","created_at":"2026-07-05T04:46:38Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"6a66b5e93d1c995c842f405113a579c687b48235b82f93e57084c1899e41c706","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2022-06-15T13:03:05Z","title_canon_sha256":"caae389baa7f3605f577773abeb775fc4feb8ba6def0be97154f306a339de040"},"schema_version":"1.0","source":{"id":"2206.07505","kind":"arxiv","version":2}},"canonical_sha256":"e6b60dc258d13c191c37d557fc62988eecb518c13ac0834717f730fe10e64347","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"e6b60dc258d13c191c37d557fc62988eecb518c13ac0834717f730fe10e64347","first_computed_at":"2026-07-05T04:46:38.453694Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T04:46:38.453694Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"KLic8Di4WAPVUBbVneIrDRPVThDaZwOdkkl3EONXZj/v16RQ1QSgUPz71WJKcnik4eApmuf8YYiC5dORO+K4CA==","signature_status":"signed_v1","signed_at":"2026-07-05T04:46:38.454134Z","signed_message":"canonical_sha256_bytes"},"source_id":"2206.07505","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:0cdc16bfb5688c96ce514a74921e27212dca0b10319e58ac0ba478e52322b1d8","sha256:b100037a6624c0c67af0adf686bbaa82e0921e8d7e3613f18a03c14cc2d0e4f7"],"state_sha256":"f1fd5c6a66075d27c591e7e3a2f731a1161ac0eab3493080d8a0c92d4b939728"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ajvZXwVb1mpYJ1K8UotGv7iqTRwXZ4f9+706DcJOmk5bdFc9OcJDPSoKRqOMU+SQCVpu03hQ9YrBxQz0xgm8Cw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-10T12:32:39.284118Z","bundle_sha256":"6b22f07788745b649f286ff4b068e54de9ebb9153f9e2bdb25d9cae5a5586bd3"}}