{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JI4DIZ224VB2Q5O3ON2YDPCZK4","short_pith_number":"pith:JI4DIZ22","schema_version":"1.0","canonical_sha256":"4a3834675ae543a875db737581bc595701e54722e934f34791e0a8c8f73bb36d","source":{"kind":"arxiv","id":"2405.19024","version":3},"attestation_state":"computed","paper":{"title":"Inverse Concave-Utility Reinforcement Learning is Inverse Game Theory","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GT","cs.MA"],"primary_cat":"cs.LG","authors_text":"Frans A. Oliehoek, Jan-Willem van de Meent, Mustafa Mert \\c{C}elikok","submitted_at":"2024-05-29T12:07:17Z","abstract_excerpt":"We consider inverse reinforcement learning problems with concave utilities. Concave Utility Reinforcement Learning (CURL) is a generalisation of the standard RL objective, which employs a concave function of the state occupancy measure, rather than a linear function. CURL has garnered recent attention for its ability to represent instances of many important applications including the standard RL such as imitation learning, pure exploration, constrained MDPs, offline RL, human-regularized RL, and others. Inverse reinforcement learning is a powerful paradigm that focuses on recovering an unknown"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.19024","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-29T12:07:17Z","cross_cats_sorted":["cs.AI","cs.GT","cs.MA"],"title_canon_sha256":"755a461adbaef16dddcdb3a24bc3cba810b17a533cc38be5161a1cfbf0e9a633","abstract_canon_sha256":"d72e54e7a1e7028a2ef637417bf15a513b8317cbd93ff5f47b0777530d0ac3c6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:51:17.697553Z","signature_b64":"AXNTl/8FjIXciif7yPvjPqzn7oRSfd7u1ASLrivKAjyWIyJKEv2M/5S2BtYv0FWr7H8B8Y5NoSATb9p8GwCNCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4a3834675ae543a875db737581bc595701e54722e934f34791e0a8c8f73bb36d","last_reissued_at":"2026-07-05T08:51:17.697015Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:51:17.697015Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Inverse Concave-Utility Reinforcement Learning is Inverse Game Theory","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GT","cs.MA"],"primary_cat":"cs.LG","authors_text":"Frans A. Oliehoek, Jan-Willem van de Meent, Mustafa Mert \\c{C}elikok","submitted_at":"2024-05-29T12:07:17Z","abstract_excerpt":"We consider inverse reinforcement learning problems with concave utilities. Concave Utility Reinforcement Learning (CURL) is a generalisation of the standard RL objective, which employs a concave function of the state occupancy measure, rather than a linear function. CURL has garnered recent attention for its ability to represent instances of many important applications including the standard RL such as imitation learning, pure exploration, constrained MDPs, offline RL, human-regularized RL, and others. Inverse reinforcement learning is a powerful paradigm that focuses on recovering an unknown"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.19024","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.19024/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.19024","created_at":"2026-07-05T08:51:17.697074+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.19024v3","created_at":"2026-07-05T08:51:17.697074+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.19024","created_at":"2026-07-05T08:51:17.697074+00:00"},{"alias_kind":"pith_short_12","alias_value":"JI4DIZ224VB2","created_at":"2026-07-05T08:51:17.697074+00:00"},{"alias_kind":"pith_short_16","alias_value":"JI4DIZ224VB2Q5O3","created_at":"2026-07-05T08:51:17.697074+00:00"},{"alias_kind":"pith_short_8","alias_value":"JI4DIZ22","created_at":"2026-07-05T08:51:17.697074+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.22292","citing_title":"Learning Incentive Structures for Cooperative Resilience in Multi-Agent Systems under Social Dilemmas","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JI4DIZ224VB2Q5O3ON2YDPCZK4","json":"https://pith.science/pith/JI4DIZ224VB2Q5O3ON2YDPCZK4.json","graph_json":"https://pith.science/api/pith-number/JI4DIZ224VB2Q5O3ON2YDPCZK4/graph.json","events_json":"https://pith.science/api/pith-number/JI4DIZ224VB2Q5O3ON2YDPCZK4/events.json","paper":"https://pith.science/paper/JI4DIZ22"},"agent_actions":{"view_html":"https://pith.science/pith/JI4DIZ224VB2Q5O3ON2YDPCZK4","download_json":"https://pith.science/pith/JI4DIZ224VB2Q5O3ON2YDPCZK4.json","view_paper":"https://pith.science/paper/JI4DIZ22","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.19024&json=true","fetch_graph":"https://pith.science/api/pith-number/JI4DIZ224VB2Q5O3ON2YDPCZK4/graph.json","fetch_events":"https://pith.science/api/pith-number/JI4DIZ224VB2Q5O3ON2YDPCZK4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JI4DIZ224VB2Q5O3ON2YDPCZK4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JI4DIZ224VB2Q5O3ON2YDPCZK4/action/storage_attestation","attest_author":"https://pith.science/pith/JI4DIZ224VB2Q5O3ON2YDPCZK4/action/author_attestation","sign_citation":"https://pith.science/pith/JI4DIZ224VB2Q5O3ON2YDPCZK4/action/citation_signature","submit_replication":"https://pith.science/pith/JI4DIZ224VB2Q5O3ON2YDPCZK4/action/replication_record"}},"created_at":"2026-07-05T08:51:17.697074+00:00","updated_at":"2026-07-05T08:51:17.697074+00:00"}