{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:YCTJNITXAGNZ4LMUTMO57535EU","short_pith_number":"pith:YCTJNITX","canonical_record":{"source":{"id":"2410.19372","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-25T08:19:49Z","cross_cats_sorted":[],"title_canon_sha256":"5798350d55fff118540c28847059ebb95d0b1fb7079fff9ece32b9880d47f0e1","abstract_canon_sha256":"0e9a439aa25068b42c7eb1d21f6462b1ad96e5f2e47a3df1ab6bd37a5b60ae7b"},"schema_version":"1.0"},"canonical_sha256":"c0a696a277019b9e2d949b1ddff77d25328fc6961d0814bda50199b326ff5dd9","source":{"kind":"arxiv","id":"2410.19372","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.19372","created_at":"2026-07-05T09:25:55Z"},{"alias_kind":"arxiv_version","alias_value":"2410.19372v1","created_at":"2026-07-05T09:25:55Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.19372","created_at":"2026-07-05T09:25:55Z"},{"alias_kind":"pith_short_12","alias_value":"YCTJNITXAGNZ","created_at":"2026-07-05T09:25:55Z"},{"alias_kind":"pith_short_16","alias_value":"YCTJNITXAGNZ4LMU","created_at":"2026-07-05T09:25:55Z"},{"alias_kind":"pith_short_8","alias_value":"YCTJNITX","created_at":"2026-07-05T09:25:55Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:YCTJNITXAGNZ4LMUTMO57535EU","target":"record","payload":{"canonical_record":{"source":{"id":"2410.19372","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-25T08:19:49Z","cross_cats_sorted":[],"title_canon_sha256":"5798350d55fff118540c28847059ebb95d0b1fb7079fff9ece32b9880d47f0e1","abstract_canon_sha256":"0e9a439aa25068b42c7eb1d21f6462b1ad96e5f2e47a3df1ab6bd37a5b60ae7b"},"schema_version":"1.0"},"canonical_sha256":"c0a696a277019b9e2d949b1ddff77d25328fc6961d0814bda50199b326ff5dd9","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:25:55.777752Z","signature_b64":"lWU70h/yY7xvOlqlijaidPzC3BZ/d5Ch/KYMljn4B4cqmgCbZ6thsdVyPhFqWQL/NWFBN8YHVpVFP0XlO1GkAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c0a696a277019b9e2d949b1ddff77d25328fc6961d0814bda50199b326ff5dd9","last_reissued_at":"2026-07-05T09:25:55.777306Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:25:55.777306Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2410.19372","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:25:55Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Y8S+/R5aJgsIxIh7aRkbf4ahXhseyymazNd1Ba+5UG3Et414RY37YlaF7ovF0h6w9uVKbJOzZfyoWOXgnb+6CA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-22T06:02:14.912755Z"},"content_sha256":"14aa2659fe0b413d343acc09d594fd5939f91b6410b225b3680fa002e70563a8","schema_version":"1.0","event_id":"sha256:14aa2659fe0b413d343acc09d594fd5939f91b6410b225b3680fa002e70563a8"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:YCTJNITXAGNZ4LMUTMO57535EU","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Toward Finding Strong Pareto Optimal Policies in Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bang Giang Le, Viet Cuong Ta","submitted_at":"2024-10-25T08:19:49Z","abstract_excerpt":"In this work, we study the problem of finding Pareto optimal policies in multi-agent reinforcement learning problems with cooperative reward structures. We show that any algorithm where each agent only optimizes their reward is subject to suboptimal convergence. Therefore, to achieve Pareto optimality, agents have to act altruistically by considering the rewards of others. This observation bridges the multi-objective optimization framework and multi-agent reinforcement learning together. We first propose a framework for applying the Multiple Gradient Descent algorithm (MGDA) for learning in mu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.19372","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.19372/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:25:55Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Rjy+kxPft+H3+hWFtzf72kPZj09zqDdiTLTh10a2o13ZsH1c17dHIpsJ4bjELPXHkizmcVS6Ii2UumDDtjOEBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-22T06:02:14.913329Z"},"content_sha256":"89ecfc99e3835d0959380e77911b6483ec2e18579c7c59ed08fb3bd5ab47e3ed","schema_version":"1.0","event_id":"sha256:89ecfc99e3835d0959380e77911b6483ec2e18579c7c59ed08fb3bd5ab47e3ed"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/YCTJNITXAGNZ4LMUTMO57535EU/bundle.json","state_url":"https://pith.science/pith/YCTJNITXAGNZ4LMUTMO57535EU/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/YCTJNITXAGNZ4LMUTMO57535EU/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-22T06:02:14Z","links":{"resolver":"https://pith.science/pith/YCTJNITXAGNZ4LMUTMO57535EU","bundle":"https://pith.science/pith/YCTJNITXAGNZ4LMUTMO57535EU/bundle.json","state":"https://pith.science/pith/YCTJNITXAGNZ4LMUTMO57535EU/state.json","well_known_bundle":"https://pith.science/.well-known/pith/YCTJNITXAGNZ4LMUTMO57535EU/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:YCTJNITXAGNZ4LMUTMO57535EU","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"0e9a439aa25068b42c7eb1d21f6462b1ad96e5f2e47a3df1ab6bd37a5b60ae7b","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-25T08:19:49Z","title_canon_sha256":"5798350d55fff118540c28847059ebb95d0b1fb7079fff9ece32b9880d47f0e1"},"schema_version":"1.0","source":{"id":"2410.19372","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.19372","created_at":"2026-07-05T09:25:55Z"},{"alias_kind":"arxiv_version","alias_value":"2410.19372v1","created_at":"2026-07-05T09:25:55Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.19372","created_at":"2026-07-05T09:25:55Z"},{"alias_kind":"pith_short_12","alias_value":"YCTJNITXAGNZ","created_at":"2026-07-05T09:25:55Z"},{"alias_kind":"pith_short_16","alias_value":"YCTJNITXAGNZ4LMU","created_at":"2026-07-05T09:25:55Z"},{"alias_kind":"pith_short_8","alias_value":"YCTJNITX","created_at":"2026-07-05T09:25:55Z"}],"graph_snapshots":[{"event_id":"sha256:89ecfc99e3835d0959380e77911b6483ec2e18579c7c59ed08fb3bd5ab47e3ed","target":"graph","created_at":"2026-07-05T09:25:55Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2410.19372/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In this work, we study the problem of finding Pareto optimal policies in multi-agent reinforcement learning problems with cooperative reward structures. We show that any algorithm where each agent only optimizes their reward is subject to suboptimal convergence. Therefore, to achieve Pareto optimality, agents have to act altruistically by considering the rewards of others. This observation bridges the multi-objective optimization framework and multi-agent reinforcement learning together. We first propose a framework for applying the Multiple Gradient Descent algorithm (MGDA) for learning in mu","authors_text":"Bang Giang Le, Viet Cuong Ta","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-25T08:19:49Z","title":"Toward Finding Strong Pareto Optimal Policies in Multi-Agent Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.19372","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:14aa2659fe0b413d343acc09d594fd5939f91b6410b225b3680fa002e70563a8","target":"record","created_at":"2026-07-05T09:25:55Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"0e9a439aa25068b42c7eb1d21f6462b1ad96e5f2e47a3df1ab6bd37a5b60ae7b","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-25T08:19:49Z","title_canon_sha256":"5798350d55fff118540c28847059ebb95d0b1fb7079fff9ece32b9880d47f0e1"},"schema_version":"1.0","source":{"id":"2410.19372","kind":"arxiv","version":1}},"canonical_sha256":"c0a696a277019b9e2d949b1ddff77d25328fc6961d0814bda50199b326ff5dd9","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"c0a696a277019b9e2d949b1ddff77d25328fc6961d0814bda50199b326ff5dd9","first_computed_at":"2026-07-05T09:25:55.777306Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:25:55.777306Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"lWU70h/yY7xvOlqlijaidPzC3BZ/d5Ch/KYMljn4B4cqmgCbZ6thsdVyPhFqWQL/NWFBN8YHVpVFP0XlO1GkAA==","signature_status":"signed_v1","signed_at":"2026-07-05T09:25:55.777752Z","signed_message":"canonical_sha256_bytes"},"source_id":"2410.19372","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:14aa2659fe0b413d343acc09d594fd5939f91b6410b225b3680fa002e70563a8","sha256:89ecfc99e3835d0959380e77911b6483ec2e18579c7c59ed08fb3bd5ab47e3ed"],"state_sha256":"b94dcbb8946e26b25ddb67f9edbdebbb52121873251d311821ea2fdb5d556739"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"He9MM3R8Smw/qRDs1hGFfhn806mCVbmlXKwgquFiXvwaVvl5DQfAZfIiLuk4XVnr8d8js8sRTqCVzdtgfHy6Dg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-22T06:02:14.918214Z","bundle_sha256":"e4efd7c1198cbe02e58e08014d10cb0d4864093f2062928e8a8b8de010886d78"}}