{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2019:SCF4JFSRQWR5VU2EP42INGUIWZ","short_pith_number":"pith:SCF4JFSR","canonical_record":{"source":{"id":"1905.06750","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-05-16T13:43:38Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"31ff2aef9835d1e2e3abc2ee1f2f958abfef3eb085967d0c61220e67c3bdf88e","abstract_canon_sha256":"da38e3af4c8e406d3a7373406b70358eaf6388c631f9e8a08f3bab877438e1c4"},"schema_version":"1.0"},"canonical_sha256":"908bc4965185a3dad3447f34869a88b64da38b4f32a4e0b91a613245bc52c466","source":{"kind":"arxiv","id":"1905.06750","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1905.06750","created_at":"2026-05-17T23:43:56Z"},{"alias_kind":"arxiv_version","alias_value":"1905.06750v2","created_at":"2026-05-17T23:43:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1905.06750","created_at":"2026-05-17T23:43:56Z"},{"alias_kind":"pith_short_12","alias_value":"SCF4JFSRQWR5","created_at":"2026-05-18T12:33:27Z"},{"alias_kind":"pith_short_16","alias_value":"SCF4JFSRQWR5VU2E","created_at":"2026-05-18T12:33:27Z"},{"alias_kind":"pith_short_8","alias_value":"SCF4JFSR","created_at":"2026-05-18T12:33:27Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2019:SCF4JFSRQWR5VU2EP42INGUIWZ","target":"record","payload":{"canonical_record":{"source":{"id":"1905.06750","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-05-16T13:43:38Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"31ff2aef9835d1e2e3abc2ee1f2f958abfef3eb085967d0c61220e67c3bdf88e","abstract_canon_sha256":"da38e3af4c8e406d3a7373406b70358eaf6388c631f9e8a08f3bab877438e1c4"},"schema_version":"1.0"},"canonical_sha256":"908bc4965185a3dad3447f34869a88b64da38b4f32a4e0b91a613245bc52c466","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-17T23:43:56.063433Z","signature_b64":"wPrrehB8TZL88AN7me6h0hm8AENC33inuwZ5TCBUkSUP1TVAtCM1F7G/WGhalrL5+VTxkrPrG9CiXsI6hYSoBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"908bc4965185a3dad3447f34869a88b64da38b4f32a4e0b91a613245bc52c466","last_reissued_at":"2026-05-17T23:43:56.062704Z","signature_status":"signed_v1","first_computed_at":"2026-05-17T23:43:56.062704Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1905.06750","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-17T23:43:56Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"usdnCV+HRVZLn4rp7k2kp54AesN+ZQEHWU81WwnE2xwDWLhITpMWWJdO4lmPJPMe943nLzF6xYFISOgK8ktPCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-25T17:08:17.479781Z"},"content_sha256":"e334f277ad6de81e6cd37ffc8840b6ca2b76591bfe7a39229e809e20dbe9cc95","schema_version":"1.0","event_id":"sha256:e334f277ad6de81e6cd37ffc8840b6ca2b76591bfe7a39229e809e20dbe9cc95"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2019:SCF4JFSRQWR5VU2EP42INGUIWZ","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Random Expert Distillation: Imitation Learning via Expert Policy Support Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Carlo Ciliberto, Pierluigi Amadori, Ruohan Wang, Yiannis Demiris","submitted_at":"2019-05-16T13:43:38Z","abstract_excerpt":"We consider the problem of imitation learning from a finite set of expert trajectories, without access to reinforcement signals. The classical approach of extracting the expert's reward function via inverse reinforcement learning, followed by reinforcement learning is indirect and may be computationally expensive. Recent generative adversarial methods based on matching the policy distribution between the expert and the agent could be unstable during training. We propose a new framework for imitation learning by estimating the support of the expert policy to compute a fixed reward function, whi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1905.06750","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-17T23:43:56Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5RPolxa0uZzA1eBQHMXb5R5/6WCpbOp/eecSrEKL3x5+RURgaxk8lHhihgOySTyjDo1h15fZ19ad4ZiJcj+oAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-25T17:08:17.480173Z"},"content_sha256":"dcc2b3f73c27be14435690c402a0b1b1fd808721cff7dc5c44fc182248a32e33","schema_version":"1.0","event_id":"sha256:dcc2b3f73c27be14435690c402a0b1b1fd808721cff7dc5c44fc182248a32e33"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/SCF4JFSRQWR5VU2EP42INGUIWZ/bundle.json","state_url":"https://pith.science/pith/SCF4JFSRQWR5VU2EP42INGUIWZ/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/SCF4JFSRQWR5VU2EP42INGUIWZ/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-05-25T17:08:17Z","links":{"resolver":"https://pith.science/pith/SCF4JFSRQWR5VU2EP42INGUIWZ","bundle":"https://pith.science/pith/SCF4JFSRQWR5VU2EP42INGUIWZ/bundle.json","state":"https://pith.science/pith/SCF4JFSRQWR5VU2EP42INGUIWZ/state.json","well_known_bundle":"https://pith.science/.well-known/pith/SCF4JFSRQWR5VU2EP42INGUIWZ/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2019:SCF4JFSRQWR5VU2EP42INGUIWZ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"da38e3af4c8e406d3a7373406b70358eaf6388c631f9e8a08f3bab877438e1c4","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-05-16T13:43:38Z","title_canon_sha256":"31ff2aef9835d1e2e3abc2ee1f2f958abfef3eb085967d0c61220e67c3bdf88e"},"schema_version":"1.0","source":{"id":"1905.06750","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1905.06750","created_at":"2026-05-17T23:43:56Z"},{"alias_kind":"arxiv_version","alias_value":"1905.06750v2","created_at":"2026-05-17T23:43:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1905.06750","created_at":"2026-05-17T23:43:56Z"},{"alias_kind":"pith_short_12","alias_value":"SCF4JFSRQWR5","created_at":"2026-05-18T12:33:27Z"},{"alias_kind":"pith_short_16","alias_value":"SCF4JFSRQWR5VU2E","created_at":"2026-05-18T12:33:27Z"},{"alias_kind":"pith_short_8","alias_value":"SCF4JFSR","created_at":"2026-05-18T12:33:27Z"}],"graph_snapshots":[{"event_id":"sha256:dcc2b3f73c27be14435690c402a0b1b1fd808721cff7dc5c44fc182248a32e33","target":"graph","created_at":"2026-05-17T23:43:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"We consider the problem of imitation learning from a finite set of expert trajectories, without access to reinforcement signals. The classical approach of extracting the expert's reward function via inverse reinforcement learning, followed by reinforcement learning is indirect and may be computationally expensive. Recent generative adversarial methods based on matching the policy distribution between the expert and the agent could be unstable during training. We propose a new framework for imitation learning by estimating the support of the expert policy to compute a fixed reward function, whi","authors_text":"Carlo Ciliberto, Pierluigi Amadori, Ruohan Wang, Yiannis Demiris","cross_cats":["stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-05-16T13:43:38Z","title":"Random Expert Distillation: Imitation Learning via Expert Policy Support Estimation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1905.06750","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:e334f277ad6de81e6cd37ffc8840b6ca2b76591bfe7a39229e809e20dbe9cc95","target":"record","created_at":"2026-05-17T23:43:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"da38e3af4c8e406d3a7373406b70358eaf6388c631f9e8a08f3bab877438e1c4","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-05-16T13:43:38Z","title_canon_sha256":"31ff2aef9835d1e2e3abc2ee1f2f958abfef3eb085967d0c61220e67c3bdf88e"},"schema_version":"1.0","source":{"id":"1905.06750","kind":"arxiv","version":2}},"canonical_sha256":"908bc4965185a3dad3447f34869a88b64da38b4f32a4e0b91a613245bc52c466","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"908bc4965185a3dad3447f34869a88b64da38b4f32a4e0b91a613245bc52c466","first_computed_at":"2026-05-17T23:43:56.062704Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-17T23:43:56.062704Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"wPrrehB8TZL88AN7me6h0hm8AENC33inuwZ5TCBUkSUP1TVAtCM1F7G/WGhalrL5+VTxkrPrG9CiXsI6hYSoBQ==","signature_status":"signed_v1","signed_at":"2026-05-17T23:43:56.063433Z","signed_message":"canonical_sha256_bytes"},"source_id":"1905.06750","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:e334f277ad6de81e6cd37ffc8840b6ca2b76591bfe7a39229e809e20dbe9cc95","sha256:dcc2b3f73c27be14435690c402a0b1b1fd808721cff7dc5c44fc182248a32e33"],"state_sha256":"b5b81823d24fc27c3869a07b1c451e9adef26076cd501a50f180455992970e23"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0l8hH5HwPOd+c6q3x6Cv4hVxby3BHFbnbqEMjHJ6W3apytlZxIMNoi2xPVVn74ca30aL3FDNpjv41x4kcSKVCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-05-25T17:08:17.483284Z","bundle_sha256":"008f7800bf932833c0fc6ace04a2bb0b9be75dc90aa8e3ee6bd4095719a7469e"}}