{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:OOIEM4CIT6HWWYUKW5ABE3NKKR","short_pith_number":"pith:OOIEM4CI","canonical_record":{"source":{"id":"2502.06261","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-10T08:53:13Z","cross_cats_sorted":[],"title_canon_sha256":"bad915f354979b0a5042a814a3d89ea5233e796e256916ae5e45b72211dd6338","abstract_canon_sha256":"1be0bcb8093cc1e289c2ba49f214ad12bff0eef5e0630f0d6677949a23ccb603"},"schema_version":"1.0"},"canonical_sha256":"73904670489f8f6b628ab740126daa5469949c7a061d8e2ac210110d8dcabd6c","source":{"kind":"arxiv","id":"2502.06261","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.06261","created_at":"2026-07-05T10:11:49Z"},{"alias_kind":"arxiv_version","alias_value":"2502.06261v1","created_at":"2026-07-05T10:11:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.06261","created_at":"2026-07-05T10:11:49Z"},{"alias_kind":"pith_short_12","alias_value":"OOIEM4CIT6HW","created_at":"2026-07-05T10:11:49Z"},{"alias_kind":"pith_short_16","alias_value":"OOIEM4CIT6HWWYUK","created_at":"2026-07-05T10:11:49Z"},{"alias_kind":"pith_short_8","alias_value":"OOIEM4CI","created_at":"2026-07-05T10:11:49Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:OOIEM4CIT6HWWYUKW5ABE3NKKR","target":"record","payload":{"canonical_record":{"source":{"id":"2502.06261","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-10T08:53:13Z","cross_cats_sorted":[],"title_canon_sha256":"bad915f354979b0a5042a814a3d89ea5233e796e256916ae5e45b72211dd6338","abstract_canon_sha256":"1be0bcb8093cc1e289c2ba49f214ad12bff0eef5e0630f0d6677949a23ccb603"},"schema_version":"1.0"},"canonical_sha256":"73904670489f8f6b628ab740126daa5469949c7a061d8e2ac210110d8dcabd6c","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:49.737353Z","signature_b64":"EJoGNb18BmoF5w1Dmr3PAs+cXZAIxaRzu4KJqeO1B+U5UztjvpSaGNqg9w53iNrcv9Sil3ehRIZUkXH4WkLyDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"73904670489f8f6b628ab740126daa5469949c7a061d8e2ac210110d8dcabd6c","last_reissued_at":"2026-07-05T10:11:49.736887Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:49.736887Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2502.06261","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:11:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+iiIWj8oFQqCsH6q90H+MQK5UyAamAnS89zXLVICb9tZImlwZLlVCQpLhd2yt44ips5X1A2lDPo6RthYPY0wAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T10:17:24.060061Z"},"content_sha256":"cedcc29bc76264c30d15dd55d009b66072c8b9414d60bf67d5353faa3f6647d8","schema_version":"1.0","event_id":"sha256:cedcc29bc76264c30d15dd55d009b66072c8b9414d60bf67d5353faa3f6647d8"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:OOIEM4CIT6HWWYUKW5ABE3NKKR","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Reducing Variance Caused by Communication in Decentralized Multi-agent Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Changxi Zhu, Mehdi Dastani, Shihan Wang","submitted_at":"2025-02-10T08:53:13Z","abstract_excerpt":"In decentralized multi-agent deep reinforcement learning (MADRL), communication can help agents to gain a better understanding of the environment to better coordinate their behaviors. Nevertheless, communication may involve uncertainty, which potentially introduces variance to the learning of decentralized agents. In this paper, we focus on a specific decentralized MADRL setting with communication and conduct a theoretical analysis to study the variance that is caused by communication in policy gradients. We propose modular techniques to reduce the variance in policy gradients during training."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.06261","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.06261/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:11:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"GcWUArdE8Ogy0kNFoSTG4D+j5mb/Musi/9GMBgzgfocmphm3gUqV9M4Qkom4VS7OLTXpYaCg83jmtKMHo5WDCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T10:17:24.060623Z"},"content_sha256":"ec1b89c1450dcfd9a59817c69c31ade577a39d0971d2b9265ab16cd96353b8fa","schema_version":"1.0","event_id":"sha256:ec1b89c1450dcfd9a59817c69c31ade577a39d0971d2b9265ab16cd96353b8fa"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/OOIEM4CIT6HWWYUKW5ABE3NKKR/bundle.json","state_url":"https://pith.science/pith/OOIEM4CIT6HWWYUKW5ABE3NKKR/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/OOIEM4CIT6HWWYUKW5ABE3NKKR/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T10:17:24Z","links":{"resolver":"https://pith.science/pith/OOIEM4CIT6HWWYUKW5ABE3NKKR","bundle":"https://pith.science/pith/OOIEM4CIT6HWWYUKW5ABE3NKKR/bundle.json","state":"https://pith.science/pith/OOIEM4CIT6HWWYUKW5ABE3NKKR/state.json","well_known_bundle":"https://pith.science/.well-known/pith/OOIEM4CIT6HWWYUKW5ABE3NKKR/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:OOIEM4CIT6HWWYUKW5ABE3NKKR","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"1be0bcb8093cc1e289c2ba49f214ad12bff0eef5e0630f0d6677949a23ccb603","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-10T08:53:13Z","title_canon_sha256":"bad915f354979b0a5042a814a3d89ea5233e796e256916ae5e45b72211dd6338"},"schema_version":"1.0","source":{"id":"2502.06261","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.06261","created_at":"2026-07-05T10:11:49Z"},{"alias_kind":"arxiv_version","alias_value":"2502.06261v1","created_at":"2026-07-05T10:11:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.06261","created_at":"2026-07-05T10:11:49Z"},{"alias_kind":"pith_short_12","alias_value":"OOIEM4CIT6HW","created_at":"2026-07-05T10:11:49Z"},{"alias_kind":"pith_short_16","alias_value":"OOIEM4CIT6HWWYUK","created_at":"2026-07-05T10:11:49Z"},{"alias_kind":"pith_short_8","alias_value":"OOIEM4CI","created_at":"2026-07-05T10:11:49Z"}],"graph_snapshots":[{"event_id":"sha256:ec1b89c1450dcfd9a59817c69c31ade577a39d0971d2b9265ab16cd96353b8fa","target":"graph","created_at":"2026-07-05T10:11:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2502.06261/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In decentralized multi-agent deep reinforcement learning (MADRL), communication can help agents to gain a better understanding of the environment to better coordinate their behaviors. Nevertheless, communication may involve uncertainty, which potentially introduces variance to the learning of decentralized agents. In this paper, we focus on a specific decentralized MADRL setting with communication and conduct a theoretical analysis to study the variance that is caused by communication in policy gradients. We propose modular techniques to reduce the variance in policy gradients during training.","authors_text":"Changxi Zhu, Mehdi Dastani, Shihan Wang","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-10T08:53:13Z","title":"Reducing Variance Caused by Communication in Decentralized Multi-agent Deep Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.06261","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:cedcc29bc76264c30d15dd55d009b66072c8b9414d60bf67d5353faa3f6647d8","target":"record","created_at":"2026-07-05T10:11:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"1be0bcb8093cc1e289c2ba49f214ad12bff0eef5e0630f0d6677949a23ccb603","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-10T08:53:13Z","title_canon_sha256":"bad915f354979b0a5042a814a3d89ea5233e796e256916ae5e45b72211dd6338"},"schema_version":"1.0","source":{"id":"2502.06261","kind":"arxiv","version":1}},"canonical_sha256":"73904670489f8f6b628ab740126daa5469949c7a061d8e2ac210110d8dcabd6c","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"73904670489f8f6b628ab740126daa5469949c7a061d8e2ac210110d8dcabd6c","first_computed_at":"2026-07-05T10:11:49.736887Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:11:49.736887Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"EJoGNb18BmoF5w1Dmr3PAs+cXZAIxaRzu4KJqeO1B+U5UztjvpSaGNqg9w53iNrcv9Sil3ehRIZUkXH4WkLyDw==","signature_status":"signed_v1","signed_at":"2026-07-05T10:11:49.737353Z","signed_message":"canonical_sha256_bytes"},"source_id":"2502.06261","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:cedcc29bc76264c30d15dd55d009b66072c8b9414d60bf67d5353faa3f6647d8","sha256:ec1b89c1450dcfd9a59817c69c31ade577a39d0971d2b9265ab16cd96353b8fa"],"state_sha256":"06628587a650e8c3abb41de872fd9a70ced0aa686a84d1f50deac931acb23276"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"VEu1M2S6jG3Y16DZksn7PzihlXdmuWO/moUCIf5eZKVCZ4vrCiVzB+GPgyE3sCqQRpqfdem8VyNQISD2sl/+Cg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T10:17:24.063660Z","bundle_sha256":"e75d568c63a643cf7f75c302bf15d0468b482dad194dd8b4554820dfaf7c92dd"}}