{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2018:AARAWT6TK3XW5TCAKTB7GA64KK","short_pith_number":"pith:AARAWT6T","canonical_record":{"source":{"id":"1806.06877","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-06-18T18:26:29Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"243c5f58785c554b62ead35059681b6ca49ba0b84c0f5a2a81e79b38e77fdd90","abstract_canon_sha256":"cbbf7b1b365995b8159d783756f67b9c86f2e5cafc1c969617cf20c95742cb19"},"schema_version":"1.0"},"canonical_sha256":"00220b4fd356ef6ecc4054c3f303dc52a2d1dba64a594cb5c0ab47b4aa62565a","source":{"kind":"arxiv","id":"1806.06877","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1806.06877","created_at":"2026-07-05T01:52:32Z"},{"alias_kind":"arxiv_version","alias_value":"1806.06877v3","created_at":"2026-07-05T01:52:32Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1806.06877","created_at":"2026-07-05T01:52:32Z"},{"alias_kind":"pith_short_12","alias_value":"AARAWT6TK3XW","created_at":"2026-07-05T01:52:32Z"},{"alias_kind":"pith_short_16","alias_value":"AARAWT6TK3XW5TCA","created_at":"2026-07-05T01:52:32Z"},{"alias_kind":"pith_short_8","alias_value":"AARAWT6T","created_at":"2026-07-05T01:52:32Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2018:AARAWT6TK3XW5TCAKTB7GA64KK","target":"record","payload":{"canonical_record":{"source":{"id":"1806.06877","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-06-18T18:26:29Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"243c5f58785c554b62ead35059681b6ca49ba0b84c0f5a2a81e79b38e77fdd90","abstract_canon_sha256":"cbbf7b1b365995b8159d783756f67b9c86f2e5cafc1c969617cf20c95742cb19"},"schema_version":"1.0"},"canonical_sha256":"00220b4fd356ef6ecc4054c3f303dc52a2d1dba64a594cb5c0ab47b4aa62565a","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:52:32.254832Z","signature_b64":"ByOat+b8dHPT9yxcJtxO53DasZhqyEii9KnGGFmPf/Fbdr3JCBcB8XG+fU5wwrlS/dA6lA4RJxJ9lusq/Z5vDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"00220b4fd356ef6ecc4054c3f303dc52a2d1dba64a594cb5c0ab47b4aa62565a","last_reissued_at":"2026-07-05T01:52:32.254371Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:52:32.254371Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1806.06877","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T01:52:32Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"aG9qeFfeCqDPMIZ2+fItJ5NcxCLV8zFUWkrFsB/wfa5aqpmszt54Hz3EryWU3j0d0kz0KPkNhB+G0VXUbYnsAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T11:25:49.328471Z"},"content_sha256":"21146eaa0c9ea296933e460629014011b797335a8b4e505b5656724a0f4f4273","schema_version":"1.0","event_id":"sha256:21146eaa0c9ea296933e460629014011b797335a8b4e505b5656724a0f4f4273"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2018:AARAWT6TK3XW5TCAKTB7GA64KK","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"A Survey of Inverse Reinforcement Learning: Challenges, Methods and Progress","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Prashant Doshi, Saurabh Arora","submitted_at":"2018-06-18T18:26:29Z","abstract_excerpt":"Inverse reinforcement learning (IRL) is the problem of inferring the reward function of an agent, given its policy or observed behavior. Analogous to RL, IRL is perceived both as a problem and as a class of methods. By categorically surveying the current literature in IRL, this article serves as a reference for researchers and practitioners of machine learning and beyond to understand the challenges of IRL and select the approaches best suited for the problem on hand. The survey formally introduces the IRL problem along with its central challenges such as the difficulty in performing accurate "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1806.06877","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1806.06877/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T01:52:32Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"1Znri0wZrQ7Uq6ApZufkreF1zGTT/VDaF3bhCNcnmYLcGw9N81tf53dvEkWlGHegP8QJO5bJ1UUy0xmz2+t6Bw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T11:25:49.329117Z"},"content_sha256":"5ce6b8cceaee72ab436ac249319b85a9991d39a18421b872a2b012bd60bf85d0","schema_version":"1.0","event_id":"sha256:5ce6b8cceaee72ab436ac249319b85a9991d39a18421b872a2b012bd60bf85d0"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/AARAWT6TK3XW5TCAKTB7GA64KK/bundle.json","state_url":"https://pith.science/pith/AARAWT6TK3XW5TCAKTB7GA64KK/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/AARAWT6TK3XW5TCAKTB7GA64KK/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-10T11:25:49Z","links":{"resolver":"https://pith.science/pith/AARAWT6TK3XW5TCAKTB7GA64KK","bundle":"https://pith.science/pith/AARAWT6TK3XW5TCAKTB7GA64KK/bundle.json","state":"https://pith.science/pith/AARAWT6TK3XW5TCAKTB7GA64KK/state.json","well_known_bundle":"https://pith.science/.well-known/pith/AARAWT6TK3XW5TCAKTB7GA64KK/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2018:AARAWT6TK3XW5TCAKTB7GA64KK","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"cbbf7b1b365995b8159d783756f67b9c86f2e5cafc1c969617cf20c95742cb19","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-06-18T18:26:29Z","title_canon_sha256":"243c5f58785c554b62ead35059681b6ca49ba0b84c0f5a2a81e79b38e77fdd90"},"schema_version":"1.0","source":{"id":"1806.06877","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1806.06877","created_at":"2026-07-05T01:52:32Z"},{"alias_kind":"arxiv_version","alias_value":"1806.06877v3","created_at":"2026-07-05T01:52:32Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1806.06877","created_at":"2026-07-05T01:52:32Z"},{"alias_kind":"pith_short_12","alias_value":"AARAWT6TK3XW","created_at":"2026-07-05T01:52:32Z"},{"alias_kind":"pith_short_16","alias_value":"AARAWT6TK3XW5TCA","created_at":"2026-07-05T01:52:32Z"},{"alias_kind":"pith_short_8","alias_value":"AARAWT6T","created_at":"2026-07-05T01:52:32Z"}],"graph_snapshots":[{"event_id":"sha256:5ce6b8cceaee72ab436ac249319b85a9991d39a18421b872a2b012bd60bf85d0","target":"graph","created_at":"2026-07-05T01:52:32Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/1806.06877/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Inverse reinforcement learning (IRL) is the problem of inferring the reward function of an agent, given its policy or observed behavior. Analogous to RL, IRL is perceived both as a problem and as a class of methods. By categorically surveying the current literature in IRL, this article serves as a reference for researchers and practitioners of machine learning and beyond to understand the challenges of IRL and select the approaches best suited for the problem on hand. The survey formally introduces the IRL problem along with its central challenges such as the difficulty in performing accurate ","authors_text":"Prashant Doshi, Saurabh Arora","cross_cats":["stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-06-18T18:26:29Z","title":"A Survey of Inverse Reinforcement Learning: Challenges, Methods and Progress"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1806.06877","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:21146eaa0c9ea296933e460629014011b797335a8b4e505b5656724a0f4f4273","target":"record","created_at":"2026-07-05T01:52:32Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"cbbf7b1b365995b8159d783756f67b9c86f2e5cafc1c969617cf20c95742cb19","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-06-18T18:26:29Z","title_canon_sha256":"243c5f58785c554b62ead35059681b6ca49ba0b84c0f5a2a81e79b38e77fdd90"},"schema_version":"1.0","source":{"id":"1806.06877","kind":"arxiv","version":3}},"canonical_sha256":"00220b4fd356ef6ecc4054c3f303dc52a2d1dba64a594cb5c0ab47b4aa62565a","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"00220b4fd356ef6ecc4054c3f303dc52a2d1dba64a594cb5c0ab47b4aa62565a","first_computed_at":"2026-07-05T01:52:32.254371Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T01:52:32.254371Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ByOat+b8dHPT9yxcJtxO53DasZhqyEii9KnGGFmPf/Fbdr3JCBcB8XG+fU5wwrlS/dA6lA4RJxJ9lusq/Z5vDQ==","signature_status":"signed_v1","signed_at":"2026-07-05T01:52:32.254832Z","signed_message":"canonical_sha256_bytes"},"source_id":"1806.06877","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:21146eaa0c9ea296933e460629014011b797335a8b4e505b5656724a0f4f4273","sha256:5ce6b8cceaee72ab436ac249319b85a9991d39a18421b872a2b012bd60bf85d0"],"state_sha256":"8bf07927f8f648a5035f1bfc230e3c40e7870c310c33f101f3f30fcad1921bdf"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"HkuOn7SHGyPCAolbDUttjAmjOuwbTC4v/M1F5IMr885bHE9lC3PALmu0Y0QZq04baT9sj/VzjH3SuWQTzt4QCQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-10T11:25:49.334908Z","bundle_sha256":"c56a39cc7c2ace4902ee33cbeaa59666902e18bdd04252887faec8a716a308d2"}}