{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:7J4XWRSDFT6BX7NHEVJPLVE7DH","short_pith_number":"pith:7J4XWRSD","canonical_record":{"source":{"id":"2510.03013","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-10-03T13:58:09Z","cross_cats_sorted":[],"title_canon_sha256":"fa091c46ada164eced69db379723da4bd2b33f4dab9996410db74c9563c8e6fd","abstract_canon_sha256":"f608370f8740778d3a53aed4e80ba9c3893ffbf091453bf623aa18487071c81e"},"schema_version":"1.0"},"canonical_sha256":"fa797b46432cfc1bfda72552f5d49f19cf328e4dd4326e928877c4e960b51235","source":{"kind":"arxiv","id":"2510.03013","version":4},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2510.03013","created_at":"2026-05-29T01:04:35Z"},{"alias_kind":"arxiv_version","alias_value":"2510.03013v4","created_at":"2026-05-29T01:04:35Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2510.03013","created_at":"2026-05-29T01:04:35Z"},{"alias_kind":"pith_short_12","alias_value":"7J4XWRSDFT6B","created_at":"2026-05-29T01:04:35Z"},{"alias_kind":"pith_short_16","alias_value":"7J4XWRSDFT6BX7NH","created_at":"2026-05-29T01:04:35Z"},{"alias_kind":"pith_short_8","alias_value":"7J4XWRSD","created_at":"2026-05-29T01:04:35Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:7J4XWRSDFT6BX7NHEVJPLVE7DH","target":"record","payload":{"canonical_record":{"source":{"id":"2510.03013","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-10-03T13:58:09Z","cross_cats_sorted":[],"title_canon_sha256":"fa091c46ada164eced69db379723da4bd2b33f4dab9996410db74c9563c8e6fd","abstract_canon_sha256":"f608370f8740778d3a53aed4e80ba9c3893ffbf091453bf623aa18487071c81e"},"schema_version":"1.0"},"canonical_sha256":"fa797b46432cfc1bfda72552f5d49f19cf328e4dd4326e928877c4e960b51235","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-29T01:04:35.845353Z","signature_b64":"aP2qNyWfEpVKnrxlGfGUakmMSFus6zE3h2OKY7pPcnTTIydKKHClZWSu59xH/sI7WE3/ccLCAWT+lqnaSPutDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fa797b46432cfc1bfda72552f5d49f19cf328e4dd4326e928877c4e960b51235","last_reissued_at":"2026-05-29T01:04:35.844781Z","signature_status":"signed_v1","first_computed_at":"2026-05-29T01:04:35.844781Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2510.03013","source_version":4,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-29T01:04:35Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4VtD0J8XsL3RyaOoqG7bP/yYtTKSy3BPsP3nl7x4ZtvzaP8E+BrMedxtNqS08JAY/QqexRxzCBypJMYpmpRHDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T16:03:17.571784Z"},"content_sha256":"b72b8717ac3bd7f53c06b4dfea7d641da80a4acfd96b04e73546351bef9c85ff","schema_version":"1.0","event_id":"sha256:b72b8717ac3bd7f53c06b4dfea7d641da80a4acfd96b04e73546351bef9c85ff"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:7J4XWRSDFT6BX7NHEVJPLVE7DH","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Distributional Inverse Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"A distributional framework for offline inverse reinforcement learning recovers full reward distributions and distribution-aware policies by minimizing first-order stochastic dominance violations.","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Anqi Wu, Feiyang Wu, Ye Zhao","submitted_at":"2025-10-03T13:58:09Z","abstract_excerpt":"We propose a distributional framework for offline Inverse Reinforcement Learning (IRL) that jointly models uncertainty over reward functions and full distributions of returns. Unlike conventional IRL approaches that recover a deterministic reward estimate or match only expected returns, our method captures richer structure in expert behavior, particularly in learning the reward distribution, by minimizing first-order stochastic dominance (FSD) violations and thus integrating distortion risk measures (DRMs) into policy learning, enabling the recovery of both reward distributions and distributio"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"our method captures richer structure in expert behavior, particularly in learning the reward distribution, by minimizing first-order stochastic dominance (FSD) violations and thus integrating distortion risk measures (DRMs) into policy learning, enabling the recovery of both reward distributions and distribution-aware policies","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The central claim rests on the premise that minimizing FSD violations is sufficient to integrate DRMs into policy learning and recover meaningful reward distributions from offline expert data without additional assumptions on the form of the return distributions or the coverage of the offline dataset.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"A distributional offline IRL method minimizes first-order stochastic dominance violations to recover reward distributions and distribution-aware policies, with O(ε^{-2}) convergence and reported SOTA results on synthetic, neurobehavioral, and MuJoCo tasks.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"A distributional framework for offline inverse reinforcement learning recovers full reward distributions and distribution-aware policies by minimizing first-order stochastic dominance violations.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"ef995ba916973ce154b9cbd314d2806f94203aee6835a19ea0db385030a0b2bd"},"source":{"id":"2510.03013","kind":"arxiv","version":4},"verdict":{"id":"e1fde4ab-e4b4-4ff7-8a1c-14e700d15d6f","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-18T10:01:54.073400Z","strongest_claim":"our method captures richer structure in expert behavior, particularly in learning the reward distribution, by minimizing first-order stochastic dominance (FSD) violations and thus integrating distortion risk measures (DRMs) into policy learning, enabling the recovery of both reward distributions and distribution-aware policies","one_line_summary":"A distributional offline IRL method minimizes first-order stochastic dominance violations to recover reward distributions and distribution-aware policies, with O(ε^{-2}) convergence and reported SOTA results on synthetic, neurobehavioral, and MuJoCo tasks.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The central claim rests on the premise that minimizing FSD violations is sufficient to integrate DRMs into policy learning and recover meaningful reward distributions from offline expert data without additional assumptions on the form of the return distributions or the coverage of the offline dataset.","pith_extraction_headline":"A distributional framework for offline inverse reinforcement learning recovers full reward distributions and distribution-aware policies by minimizing first-order stochastic dominance violations."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2510.03013/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":2,"snapshot_sha256":"220a56f344975ba1f3b28749fb4952152bf7243839378b597099348821cdd44f"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"e1fde4ab-e4b4-4ff7-8a1c-14e700d15d6f"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-29T01:04:35Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"C/OHd65hfhgcHPABKPbqUoiObZ0Eju5HQWhLOWIxbR1jZMEYqrVrezCtdoCWANfE/fCtqtZL6fYydWRV3J5lAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T16:03:17.572518Z"},"content_sha256":"da493988350ff32b52849445e48fbc835694230aecaba7020c2af1d03e5cb920","schema_version":"1.0","event_id":"sha256:da493988350ff32b52849445e48fbc835694230aecaba7020c2af1d03e5cb920"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/7J4XWRSDFT6BX7NHEVJPLVE7DH/bundle.json","state_url":"https://pith.science/pith/7J4XWRSDFT6BX7NHEVJPLVE7DH/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/7J4XWRSDFT6BX7NHEVJPLVE7DH/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T16:03:17Z","links":{"resolver":"https://pith.science/pith/7J4XWRSDFT6BX7NHEVJPLVE7DH","bundle":"https://pith.science/pith/7J4XWRSDFT6BX7NHEVJPLVE7DH/bundle.json","state":"https://pith.science/pith/7J4XWRSDFT6BX7NHEVJPLVE7DH/state.json","well_known_bundle":"https://pith.science/.well-known/pith/7J4XWRSDFT6BX7NHEVJPLVE7DH/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:7J4XWRSDFT6BX7NHEVJPLVE7DH","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f608370f8740778d3a53aed4e80ba9c3893ffbf091453bf623aa18487071c81e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-10-03T13:58:09Z","title_canon_sha256":"fa091c46ada164eced69db379723da4bd2b33f4dab9996410db74c9563c8e6fd"},"schema_version":"1.0","source":{"id":"2510.03013","kind":"arxiv","version":4}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2510.03013","created_at":"2026-05-29T01:04:35Z"},{"alias_kind":"arxiv_version","alias_value":"2510.03013v4","created_at":"2026-05-29T01:04:35Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2510.03013","created_at":"2026-05-29T01:04:35Z"},{"alias_kind":"pith_short_12","alias_value":"7J4XWRSDFT6B","created_at":"2026-05-29T01:04:35Z"},{"alias_kind":"pith_short_16","alias_value":"7J4XWRSDFT6BX7NH","created_at":"2026-05-29T01:04:35Z"},{"alias_kind":"pith_short_8","alias_value":"7J4XWRSD","created_at":"2026-05-29T01:04:35Z"}],"graph_snapshots":[{"event_id":"sha256:da493988350ff32b52849445e48fbc835694230aecaba7020c2af1d03e5cb920","target":"graph","created_at":"2026-05-29T01:04:35Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"our method captures richer structure in expert behavior, particularly in learning the reward distribution, by minimizing first-order stochastic dominance (FSD) violations and thus integrating distortion risk measures (DRMs) into policy learning, enabling the recovery of both reward distributions and distribution-aware policies"},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The central claim rests on the premise that minimizing FSD violations is sufficient to integrate DRMs into policy learning and recover meaningful reward distributions from offline expert data without additional assumptions on the form of the return distributions or the coverage of the offline dataset."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"A distributional offline IRL method minimizes first-order stochastic dominance violations to recover reward distributions and distribution-aware policies, with O(ε^{-2}) convergence and reported SOTA results on synthetic, neurobehavioral, and MuJoCo tasks."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"A distributional framework for offline inverse reinforcement learning recovers full reward distributions and distribution-aware policies by minimizing first-order stochastic dominance violations."}],"snapshot_sha256":"ef995ba916973ce154b9cbd314d2806f94203aee6835a19ea0db385030a0b2bd"},"formal_canon":{"evidence_count":2,"snapshot_sha256":"220a56f344975ba1f3b28749fb4952152bf7243839378b597099348821cdd44f"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2510.03013/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We propose a distributional framework for offline Inverse Reinforcement Learning (IRL) that jointly models uncertainty over reward functions and full distributions of returns. Unlike conventional IRL approaches that recover a deterministic reward estimate or match only expected returns, our method captures richer structure in expert behavior, particularly in learning the reward distribution, by minimizing first-order stochastic dominance (FSD) violations and thus integrating distortion risk measures (DRMs) into policy learning, enabling the recovery of both reward distributions and distributio","authors_text":"Anqi Wu, Feiyang Wu, Ye Zhao","cross_cats":[],"headline":"A distributional framework for offline inverse reinforcement learning recovers full reward distributions and distribution-aware policies by minimizing first-order stochastic dominance violations.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-10-03T13:58:09Z","title":"Distributional Inverse Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2510.03013","kind":"arxiv","version":4},"verdict":{"created_at":"2026-05-18T10:01:54.073400Z","id":"e1fde4ab-e4b4-4ff7-8a1c-14e700d15d6f","model_set":{"reader":"grok-4.3"},"one_line_summary":"A distributional offline IRL method minimizes first-order stochastic dominance violations to recover reward distributions and distribution-aware policies, with O(ε^{-2}) convergence and reported SOTA results on synthetic, neurobehavioral, and MuJoCo tasks.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"A distributional framework for offline inverse reinforcement learning recovers full reward distributions and distribution-aware policies by minimizing first-order stochastic dominance violations.","strongest_claim":"our method captures richer structure in expert behavior, particularly in learning the reward distribution, by minimizing first-order stochastic dominance (FSD) violations and thus integrating distortion risk measures (DRMs) into policy learning, enabling the recovery of both reward distributions and distribution-aware policies","weakest_assumption":"The central claim rests on the premise that minimizing FSD violations is sufficient to integrate DRMs into policy learning and recover meaningful reward distributions from offline expert data without additional assumptions on the form of the return distributions or the coverage of the offline dataset."}},"verdict_id":"e1fde4ab-e4b4-4ff7-8a1c-14e700d15d6f"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:b72b8717ac3bd7f53c06b4dfea7d641da80a4acfd96b04e73546351bef9c85ff","target":"record","created_at":"2026-05-29T01:04:35Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f608370f8740778d3a53aed4e80ba9c3893ffbf091453bf623aa18487071c81e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-10-03T13:58:09Z","title_canon_sha256":"fa091c46ada164eced69db379723da4bd2b33f4dab9996410db74c9563c8e6fd"},"schema_version":"1.0","source":{"id":"2510.03013","kind":"arxiv","version":4}},"canonical_sha256":"fa797b46432cfc1bfda72552f5d49f19cf328e4dd4326e928877c4e960b51235","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"fa797b46432cfc1bfda72552f5d49f19cf328e4dd4326e928877c4e960b51235","first_computed_at":"2026-05-29T01:04:35.844781Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-29T01:04:35.844781Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"aP2qNyWfEpVKnrxlGfGUakmMSFus6zE3h2OKY7pPcnTTIydKKHClZWSu59xH/sI7WE3/ccLCAWT+lqnaSPutDg==","signature_status":"signed_v1","signed_at":"2026-05-29T01:04:35.845353Z","signed_message":"canonical_sha256_bytes"},"source_id":"2510.03013","source_kind":"arxiv","source_version":4}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:b72b8717ac3bd7f53c06b4dfea7d641da80a4acfd96b04e73546351bef9c85ff","sha256:da493988350ff32b52849445e48fbc835694230aecaba7020c2af1d03e5cb920"],"state_sha256":"06525f1d98ac2c1e12aa52d6df0488b86384f0ad084112e52618835047a7818d"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"dXaAQD5V1is6dUM7+HmVTpMr1FVaZmt7q1Z8Tk1j3Tw0U1O2Ai8Qt7zOkC0Ls10wYmmG2zqwJieHmcx0JhnoCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T16:03:17.577245Z","bundle_sha256":"9937f7207b562fdf32993b499cbd1f9e8e10b8c60217a1851bf95d3aa2553eba"}}