{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YCTJNITXAGNZ4LMUTMO57535EU","short_pith_number":"pith:YCTJNITX","schema_version":"1.0","canonical_sha256":"c0a696a277019b9e2d949b1ddff77d25328fc6961d0814bda50199b326ff5dd9","source":{"kind":"arxiv","id":"2410.19372","version":1},"attestation_state":"computed","paper":{"title":"Toward Finding Strong Pareto Optimal Policies in Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bang Giang Le, Viet Cuong Ta","submitted_at":"2024-10-25T08:19:49Z","abstract_excerpt":"In this work, we study the problem of finding Pareto optimal policies in multi-agent reinforcement learning problems with cooperative reward structures. We show that any algorithm where each agent only optimizes their reward is subject to suboptimal convergence. Therefore, to achieve Pareto optimality, agents have to act altruistically by considering the rewards of others. This observation bridges the multi-objective optimization framework and multi-agent reinforcement learning together. We first propose a framework for applying the Multiple Gradient Descent algorithm (MGDA) for learning in mu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.19372","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-25T08:19:49Z","cross_cats_sorted":[],"title_canon_sha256":"5798350d55fff118540c28847059ebb95d0b1fb7079fff9ece32b9880d47f0e1","abstract_canon_sha256":"0e9a439aa25068b42c7eb1d21f6462b1ad96e5f2e47a3df1ab6bd37a5b60ae7b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:25:55.777752Z","signature_b64":"lWU70h/yY7xvOlqlijaidPzC3BZ/d5Ch/KYMljn4B4cqmgCbZ6thsdVyPhFqWQL/NWFBN8YHVpVFP0XlO1GkAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c0a696a277019b9e2d949b1ddff77d25328fc6961d0814bda50199b326ff5dd9","last_reissued_at":"2026-07-05T09:25:55.777306Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:25:55.777306Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Toward Finding Strong Pareto Optimal Policies in Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bang Giang Le, Viet Cuong Ta","submitted_at":"2024-10-25T08:19:49Z","abstract_excerpt":"In this work, we study the problem of finding Pareto optimal policies in multi-agent reinforcement learning problems with cooperative reward structures. We show that any algorithm where each agent only optimizes their reward is subject to suboptimal convergence. Therefore, to achieve Pareto optimality, agents have to act altruistically by considering the rewards of others. This observation bridges the multi-objective optimization framework and multi-agent reinforcement learning together. We first propose a framework for applying the Multiple Gradient Descent algorithm (MGDA) for learning in mu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.19372","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.19372/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.19372","created_at":"2026-07-05T09:25:55.777383+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.19372v1","created_at":"2026-07-05T09:25:55.777383+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.19372","created_at":"2026-07-05T09:25:55.777383+00:00"},{"alias_kind":"pith_short_12","alias_value":"YCTJNITXAGNZ","created_at":"2026-07-05T09:25:55.777383+00:00"},{"alias_kind":"pith_short_16","alias_value":"YCTJNITXAGNZ4LMU","created_at":"2026-07-05T09:25:55.777383+00:00"},{"alias_kind":"pith_short_8","alias_value":"YCTJNITX","created_at":"2026-07-05T09:25:55.777383+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YCTJNITXAGNZ4LMUTMO57535EU","json":"https://pith.science/pith/YCTJNITXAGNZ4LMUTMO57535EU.json","graph_json":"https://pith.science/api/pith-number/YCTJNITXAGNZ4LMUTMO57535EU/graph.json","events_json":"https://pith.science/api/pith-number/YCTJNITXAGNZ4LMUTMO57535EU/events.json","paper":"https://pith.science/paper/YCTJNITX"},"agent_actions":{"view_html":"https://pith.science/pith/YCTJNITXAGNZ4LMUTMO57535EU","download_json":"https://pith.science/pith/YCTJNITXAGNZ4LMUTMO57535EU.json","view_paper":"https://pith.science/paper/YCTJNITX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.19372&json=true","fetch_graph":"https://pith.science/api/pith-number/YCTJNITXAGNZ4LMUTMO57535EU/graph.json","fetch_events":"https://pith.science/api/pith-number/YCTJNITXAGNZ4LMUTMO57535EU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YCTJNITXAGNZ4LMUTMO57535EU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YCTJNITXAGNZ4LMUTMO57535EU/action/storage_attestation","attest_author":"https://pith.science/pith/YCTJNITXAGNZ4LMUTMO57535EU/action/author_attestation","sign_citation":"https://pith.science/pith/YCTJNITXAGNZ4LMUTMO57535EU/action/citation_signature","submit_replication":"https://pith.science/pith/YCTJNITXAGNZ4LMUTMO57535EU/action/replication_record"}},"created_at":"2026-07-05T09:25:55.777383+00:00","updated_at":"2026-07-05T09:25:55.777383+00:00"}