{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:A7ANV5TCL4ZPP7E2TPVJKMD3FR","short_pith_number":"pith:A7ANV5TC","schema_version":"1.0","canonical_sha256":"07c0daf6625f32f7fc9a9bea95307b2c4ed32eddf120049258a3168f3446ec73","source":{"kind":"arxiv","id":"2208.12584","version":2},"attestation_state":"computed","paper":{"title":"Socially Fair Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CY","cs.GT","cs.MA"],"primary_cat":"cs.LG","authors_text":"Debmalya Mandal, Jiarui Gan","submitted_at":"2022-08-26T11:01:55Z","abstract_excerpt":"We consider the problem of episodic reinforcement learning where there are multiple stakeholders with different reward functions. Our goal is to output a policy that is socially fair with respect to different reward functions. Prior works have proposed different objectives that a fair policy must optimize including minimum welfare, and generalized Gini welfare. We first take an axiomatic view of the problem, and propose four axioms that any such fair objective must satisfy. We show that the Nash social welfare is the unique objective that uniquely satisfies all four objectives, whereas prior o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2208.12584","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-08-26T11:01:55Z","cross_cats_sorted":["cs.CY","cs.GT","cs.MA"],"title_canon_sha256":"50db9c6bb2524740bdff7c9173d73ad2840c6ab8c5503061b0ae79047c7196a6","abstract_canon_sha256":"8015b1bfb0e5cd6a998754a66c673e5f80f2d8d8c91174d739c7ff81b9428498"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:38:28.770363Z","signature_b64":"egxT3lhK4NRbW3kZwfLA1kTUgSr/f1jvs4Og+YsLvp+dSIYOlPsgoz28+3RDlQLTQOwiId8Wrs99INd0c875BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"07c0daf6625f32f7fc9a9bea95307b2c4ed32eddf120049258a3168f3446ec73","last_reissued_at":"2026-07-05T05:38:28.769928Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:38:28.769928Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Socially Fair Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CY","cs.GT","cs.MA"],"primary_cat":"cs.LG","authors_text":"Debmalya Mandal, Jiarui Gan","submitted_at":"2022-08-26T11:01:55Z","abstract_excerpt":"We consider the problem of episodic reinforcement learning where there are multiple stakeholders with different reward functions. Our goal is to output a policy that is socially fair with respect to different reward functions. Prior works have proposed different objectives that a fair policy must optimize including minimum welfare, and generalized Gini welfare. We first take an axiomatic view of the problem, and propose four axioms that any such fair objective must satisfy. We show that the Nash social welfare is the unique objective that uniquely satisfies all four objectives, whereas prior o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2208.12584","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2208.12584/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2208.12584","created_at":"2026-07-05T05:38:28.769982+00:00"},{"alias_kind":"arxiv_version","alias_value":"2208.12584v2","created_at":"2026-07-05T05:38:28.769982+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2208.12584","created_at":"2026-07-05T05:38:28.769982+00:00"},{"alias_kind":"pith_short_12","alias_value":"A7ANV5TCL4ZP","created_at":"2026-07-05T05:38:28.769982+00:00"},{"alias_kind":"pith_short_16","alias_value":"A7ANV5TCL4ZPP7E2","created_at":"2026-07-05T05:38:28.769982+00:00"},{"alias_kind":"pith_short_8","alias_value":"A7ANV5TC","created_at":"2026-07-05T05:38:28.769982+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23931","citing_title":"Welfarist Control Design -- How to fulfill the societal mandate in multi-agent control?","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18111","citing_title":"Learning Fair Pareto-Optimal Policies in Multi-Objective Reinforcement Learning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01961","citing_title":"Multi-User Dueling Bandits: A Fair Approach using Nash Social Welfare","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A7ANV5TCL4ZPP7E2TPVJKMD3FR","json":"https://pith.science/pith/A7ANV5TCL4ZPP7E2TPVJKMD3FR.json","graph_json":"https://pith.science/api/pith-number/A7ANV5TCL4ZPP7E2TPVJKMD3FR/graph.json","events_json":"https://pith.science/api/pith-number/A7ANV5TCL4ZPP7E2TPVJKMD3FR/events.json","paper":"https://pith.science/paper/A7ANV5TC"},"agent_actions":{"view_html":"https://pith.science/pith/A7ANV5TCL4ZPP7E2TPVJKMD3FR","download_json":"https://pith.science/pith/A7ANV5TCL4ZPP7E2TPVJKMD3FR.json","view_paper":"https://pith.science/paper/A7ANV5TC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2208.12584&json=true","fetch_graph":"https://pith.science/api/pith-number/A7ANV5TCL4ZPP7E2TPVJKMD3FR/graph.json","fetch_events":"https://pith.science/api/pith-number/A7ANV5TCL4ZPP7E2TPVJKMD3FR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A7ANV5TCL4ZPP7E2TPVJKMD3FR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A7ANV5TCL4ZPP7E2TPVJKMD3FR/action/storage_attestation","attest_author":"https://pith.science/pith/A7ANV5TCL4ZPP7E2TPVJKMD3FR/action/author_attestation","sign_citation":"https://pith.science/pith/A7ANV5TCL4ZPP7E2TPVJKMD3FR/action/citation_signature","submit_replication":"https://pith.science/pith/A7ANV5TCL4ZPP7E2TPVJKMD3FR/action/replication_record"}},"created_at":"2026-07-05T05:38:28.769982+00:00","updated_at":"2026-07-05T05:38:28.769982+00:00"}