{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2020:X6ZITYDXED27FTLUES3ULNPZSV","short_pith_number":"pith:X6ZITYDX","canonical_record":{"source":{"id":"2012.06899","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-12-12T20:06:15Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"1f437af52266d6d6f2beca1e44d2e3a0d58b34d001e012f30d586f0dc3e0c50b","abstract_canon_sha256":"cb94d115b102f15fb6a2053a14865920fe9a1c35f4a5998aa891e5f505544544"},"schema_version":"1.0"},"canonical_sha256":"bfb289e07720f5f2cd7424b745b5f9955de5f882bd8691c58a3e297ffbd67c33","source":{"kind":"arxiv","id":"2012.06899","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2012.06899","created_at":"2026-07-05T01:59:00Z"},{"alias_kind":"arxiv_version","alias_value":"2012.06899v1","created_at":"2026-07-05T01:59:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.06899","created_at":"2026-07-05T01:59:00Z"},{"alias_kind":"pith_short_12","alias_value":"X6ZITYDXED27","created_at":"2026-07-05T01:59:00Z"},{"alias_kind":"pith_short_16","alias_value":"X6ZITYDXED27FTLU","created_at":"2026-07-05T01:59:00Z"},{"alias_kind":"pith_short_8","alias_value":"X6ZITYDX","created_at":"2026-07-05T01:59:00Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2020:X6ZITYDXED27FTLUES3ULNPZSV","target":"record","payload":{"canonical_record":{"source":{"id":"2012.06899","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-12-12T20:06:15Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"1f437af52266d6d6f2beca1e44d2e3a0d58b34d001e012f30d586f0dc3e0c50b","abstract_canon_sha256":"cb94d115b102f15fb6a2053a14865920fe9a1c35f4a5998aa891e5f505544544"},"schema_version":"1.0"},"canonical_sha256":"bfb289e07720f5f2cd7424b745b5f9955de5f882bd8691c58a3e297ffbd67c33","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:59:00.991906Z","signature_b64":"lMjNTbuRjaoTnxty95t4K/VY1Sj0nK2/jmsZKlFOzjSMZTDlrRx/WpNKjObY/jm1dTJpacsUQwAI4O8QRXeUCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bfb289e07720f5f2cd7424b745b5f9955de5f882bd8691c58a3e297ffbd67c33","last_reissued_at":"2026-07-05T01:59:00.991423Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:59:00.991423Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2012.06899","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T01:59:00Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"dFfXN3jfSInItbjtZ3BiC1hA2vAjy++g8mF6gvuKiojIXLwO0J5S5Ji95dQ38kG+//D6I2pqVUNgNZd6KUJ9BA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T17:13:35.255316Z"},"content_sha256":"64628cde257202bdc884065ff58092b396762de23fb70cdd6d5268d154a40de7","schema_version":"1.0","event_id":"sha256:64628cde257202bdc884065ff58092b396762de23fb70cdd6d5268d154a40de7"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2020:X6ZITYDXED27FTLUES3ULNPZSV","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Semi-supervised reward learning for offline reinforcement learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Alexander Novikov, Konrad Zolna, Ksenia Konyushkova, Nando de Freitas, Scott Reed, Serkan Cabi, Yusuf Aytar","submitted_at":"2020-12-12T20:06:15Z","abstract_excerpt":"In offline reinforcement learning (RL) agents are trained using a logged dataset. It appears to be the most natural route to attack real-life applications because in domains such as healthcare and robotics interactions with the environment are either expensive or unethical. Training agents usually requires reward functions, but unfortunately, rewards are seldom available in practice and their engineering is challenging and laborious. To overcome this, we investigate reward learning under the constraint of minimizing human reward annotations. We consider two types of supervision: timestep annot"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.06899","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.06899/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T01:59:00Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"bHJYMd8b9S58rngw+nCLShTTDwdXs1E6N/7JoJU++MZo6yyc5NJh4vQEAmwmpwaz4BaR7/8RaGBY5q4LnbzADQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T17:13:35.255822Z"},"content_sha256":"1d98aab7e5b9660b117fc704081376c7c66a28749f3f7f609a0fd6bd8ab900e2","schema_version":"1.0","event_id":"sha256:1d98aab7e5b9660b117fc704081376c7c66a28749f3f7f609a0fd6bd8ab900e2"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/X6ZITYDXED27FTLUES3ULNPZSV/bundle.json","state_url":"https://pith.science/pith/X6ZITYDXED27FTLUES3ULNPZSV/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/X6ZITYDXED27FTLUES3ULNPZSV/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T17:13:35Z","links":{"resolver":"https://pith.science/pith/X6ZITYDXED27FTLUES3ULNPZSV","bundle":"https://pith.science/pith/X6ZITYDXED27FTLUES3ULNPZSV/bundle.json","state":"https://pith.science/pith/X6ZITYDXED27FTLUES3ULNPZSV/state.json","well_known_bundle":"https://pith.science/.well-known/pith/X6ZITYDXED27FTLUES3ULNPZSV/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:X6ZITYDXED27FTLUES3ULNPZSV","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"cb94d115b102f15fb6a2053a14865920fe9a1c35f4a5998aa891e5f505544544","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-12-12T20:06:15Z","title_canon_sha256":"1f437af52266d6d6f2beca1e44d2e3a0d58b34d001e012f30d586f0dc3e0c50b"},"schema_version":"1.0","source":{"id":"2012.06899","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2012.06899","created_at":"2026-07-05T01:59:00Z"},{"alias_kind":"arxiv_version","alias_value":"2012.06899v1","created_at":"2026-07-05T01:59:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.06899","created_at":"2026-07-05T01:59:00Z"},{"alias_kind":"pith_short_12","alias_value":"X6ZITYDXED27","created_at":"2026-07-05T01:59:00Z"},{"alias_kind":"pith_short_16","alias_value":"X6ZITYDXED27FTLU","created_at":"2026-07-05T01:59:00Z"},{"alias_kind":"pith_short_8","alias_value":"X6ZITYDX","created_at":"2026-07-05T01:59:00Z"}],"graph_snapshots":[{"event_id":"sha256:1d98aab7e5b9660b117fc704081376c7c66a28749f3f7f609a0fd6bd8ab900e2","target":"graph","created_at":"2026-07-05T01:59:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2012.06899/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In offline reinforcement learning (RL) agents are trained using a logged dataset. It appears to be the most natural route to attack real-life applications because in domains such as healthcare and robotics interactions with the environment are either expensive or unethical. Training agents usually requires reward functions, but unfortunately, rewards are seldom available in practice and their engineering is challenging and laborious. To overcome this, we investigate reward learning under the constraint of minimizing human reward annotations. We consider two types of supervision: timestep annot","authors_text":"Alexander Novikov, Konrad Zolna, Ksenia Konyushkova, Nando de Freitas, Scott Reed, Serkan Cabi, Yusuf Aytar","cross_cats":["cs.AI","cs.RO"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-12-12T20:06:15Z","title":"Semi-supervised reward learning for offline reinforcement learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.06899","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:64628cde257202bdc884065ff58092b396762de23fb70cdd6d5268d154a40de7","target":"record","created_at":"2026-07-05T01:59:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"cb94d115b102f15fb6a2053a14865920fe9a1c35f4a5998aa891e5f505544544","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-12-12T20:06:15Z","title_canon_sha256":"1f437af52266d6d6f2beca1e44d2e3a0d58b34d001e012f30d586f0dc3e0c50b"},"schema_version":"1.0","source":{"id":"2012.06899","kind":"arxiv","version":1}},"canonical_sha256":"bfb289e07720f5f2cd7424b745b5f9955de5f882bd8691c58a3e297ffbd67c33","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"bfb289e07720f5f2cd7424b745b5f9955de5f882bd8691c58a3e297ffbd67c33","first_computed_at":"2026-07-05T01:59:00.991423Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T01:59:00.991423Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"lMjNTbuRjaoTnxty95t4K/VY1Sj0nK2/jmsZKlFOzjSMZTDlrRx/WpNKjObY/jm1dTJpacsUQwAI4O8QRXeUCw==","signature_status":"signed_v1","signed_at":"2026-07-05T01:59:00.991906Z","signed_message":"canonical_sha256_bytes"},"source_id":"2012.06899","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:64628cde257202bdc884065ff58092b396762de23fb70cdd6d5268d154a40de7","sha256:1d98aab7e5b9660b117fc704081376c7c66a28749f3f7f609a0fd6bd8ab900e2"],"state_sha256":"fd445a1e2162d7524fac86a1c46429a7bb717e3c4085896ed5274baaea9940ad"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"QH+TrvHflOt5LDGhXy70C8nmxqVLhGKzXyNxZn3+jjoI+VMOBl37kHB4qSWPq1q9pGvPX5a3HCO5USBoy6JABg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T17:13:35.260366Z","bundle_sha256":"6fbe4e97873f46e94bd49332ced9488fa99e7d2d596be0d92e202d7e71f27515"}}