{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:3KCR72CC65QWVETGDIJVL4QZR2","short_pith_number":"pith:3KCR72CC","canonical_record":{"source":{"id":"2106.06499","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-11T16:49:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"75829c55e8ef67d02ada77127a5d56bd41806e7fca5e4a9ec60633e4f9b4c130","abstract_canon_sha256":"11b3bf01738284673610c66cda3097c04b35239f7a09c46fa6ec46853197d7f1"},"schema_version":"1.0"},"canonical_sha256":"da851fe842f7616a92661a1355f2198e9ce3e2a02848c02611542f361f0ea62c","source":{"kind":"arxiv","id":"2106.06499","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2106.06499","created_at":"2026-07-05T02:51:27Z"},{"alias_kind":"arxiv_version","alias_value":"2106.06499v2","created_at":"2026-07-05T02:51:27Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.06499","created_at":"2026-07-05T02:51:27Z"},{"alias_kind":"pith_short_12","alias_value":"3KCR72CC65QW","created_at":"2026-07-05T02:51:27Z"},{"alias_kind":"pith_short_16","alias_value":"3KCR72CC65QWVETG","created_at":"2026-07-05T02:51:27Z"},{"alias_kind":"pith_short_8","alias_value":"3KCR72CC","created_at":"2026-07-05T02:51:27Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:3KCR72CC65QWVETGDIJVL4QZR2","target":"record","payload":{"canonical_record":{"source":{"id":"2106.06499","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-11T16:49:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"75829c55e8ef67d02ada77127a5d56bd41806e7fca5e4a9ec60633e4f9b4c130","abstract_canon_sha256":"11b3bf01738284673610c66cda3097c04b35239f7a09c46fa6ec46853197d7f1"},"schema_version":"1.0"},"canonical_sha256":"da851fe842f7616a92661a1355f2198e9ce3e2a02848c02611542f361f0ea62c","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:51:27.300797Z","signature_b64":"519L3e4jtpEod9F3MvdVFP7NVIzDFf6bPHg8qKV9fvB1vcJoHI0MBl2/SUFmflgZStJdt89ZH7jVqFfVUHV5DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"da851fe842f7616a92661a1355f2198e9ce3e2a02848c02611542f361f0ea62c","last_reissued_at":"2026-07-05T02:51:27.300236Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:51:27.300236Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2106.06499","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:51:27Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"rxs2YK2LpBJN//ow2OotVYdxOhQzJkPh7NbPbsm5sn4Wz7+HYknhd0i0CRUpKMdm0P7XUq+WC2tdvBgNtjh1Cw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T15:53:27.297567Z"},"content_sha256":"ccc21d46f2d0c69be5c5cd59477516d916bd16b6353fc21a2c264c315f9c7e6a","schema_version":"1.0","event_id":"sha256:ccc21d46f2d0c69be5c5cd59477516d916bd16b6353fc21a2c264c315f9c7e6a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:3KCR72CC65QWVETGDIJVL4QZR2","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Policy Gradient Bayesian Robust Optimization for Imitation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Anca D. Dragan, Ashwin Balakrishna, Daniel S. Brown, Jerry Zhu, Ken Goldberg, Marek Petrik, Satvik Sharma, Zaynah Javed","submitted_at":"2021-06-11T16:49:15Z","abstract_excerpt":"The difficulty in specifying rewards for many real-world problems has led to an increased focus on learning rewards from human feedback, such as demonstrations. However, there are often many different reward functions that explain the human feedback, leaving agents with uncertainty over what the true reward function is. While most policy optimization approaches handle this uncertainty by optimizing for expected performance, many applications demand risk-averse behavior. We derive a novel policy gradient-style robust optimization approach, PG-BROIL, that optimizes a soft-robust objective that b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.06499","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.06499/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:51:27Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"jl5PsuawnWChFFLv0vDwg+i7Ifo5mMkn6vbCwehqYrtMCQRG9wH+i6+tVUn8TNY7nukiFswiXm+Cn8pmdP8cDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T15:53:27.298064Z"},"content_sha256":"bd79a69ce9fc87d212a5bfe2dec0ab80981c1f91dac56aeece01e58b48733bc3","schema_version":"1.0","event_id":"sha256:bd79a69ce9fc87d212a5bfe2dec0ab80981c1f91dac56aeece01e58b48733bc3"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/3KCR72CC65QWVETGDIJVL4QZR2/bundle.json","state_url":"https://pith.science/pith/3KCR72CC65QWVETGDIJVL4QZR2/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/3KCR72CC65QWVETGDIJVL4QZR2/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-03T15:53:27Z","links":{"resolver":"https://pith.science/pith/3KCR72CC65QWVETGDIJVL4QZR2","bundle":"https://pith.science/pith/3KCR72CC65QWVETGDIJVL4QZR2/bundle.json","state":"https://pith.science/pith/3KCR72CC65QWVETGDIJVL4QZR2/state.json","well_known_bundle":"https://pith.science/.well-known/pith/3KCR72CC65QWVETGDIJVL4QZR2/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:3KCR72CC65QWVETGDIJVL4QZR2","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"11b3bf01738284673610c66cda3097c04b35239f7a09c46fa6ec46853197d7f1","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-11T16:49:15Z","title_canon_sha256":"75829c55e8ef67d02ada77127a5d56bd41806e7fca5e4a9ec60633e4f9b4c130"},"schema_version":"1.0","source":{"id":"2106.06499","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2106.06499","created_at":"2026-07-05T02:51:27Z"},{"alias_kind":"arxiv_version","alias_value":"2106.06499v2","created_at":"2026-07-05T02:51:27Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.06499","created_at":"2026-07-05T02:51:27Z"},{"alias_kind":"pith_short_12","alias_value":"3KCR72CC65QW","created_at":"2026-07-05T02:51:27Z"},{"alias_kind":"pith_short_16","alias_value":"3KCR72CC65QWVETG","created_at":"2026-07-05T02:51:27Z"},{"alias_kind":"pith_short_8","alias_value":"3KCR72CC","created_at":"2026-07-05T02:51:27Z"}],"graph_snapshots":[{"event_id":"sha256:bd79a69ce9fc87d212a5bfe2dec0ab80981c1f91dac56aeece01e58b48733bc3","target":"graph","created_at":"2026-07-05T02:51:27Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2106.06499/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The difficulty in specifying rewards for many real-world problems has led to an increased focus on learning rewards from human feedback, such as demonstrations. However, there are often many different reward functions that explain the human feedback, leaving agents with uncertainty over what the true reward function is. While most policy optimization approaches handle this uncertainty by optimizing for expected performance, many applications demand risk-averse behavior. We derive a novel policy gradient-style robust optimization approach, PG-BROIL, that optimizes a soft-robust objective that b","authors_text":"Anca D. Dragan, Ashwin Balakrishna, Daniel S. Brown, Jerry Zhu, Ken Goldberg, Marek Petrik, Satvik Sharma, Zaynah Javed","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-11T16:49:15Z","title":"Policy Gradient Bayesian Robust Optimization for Imitation Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.06499","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ccc21d46f2d0c69be5c5cd59477516d916bd16b6353fc21a2c264c315f9c7e6a","target":"record","created_at":"2026-07-05T02:51:27Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"11b3bf01738284673610c66cda3097c04b35239f7a09c46fa6ec46853197d7f1","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-11T16:49:15Z","title_canon_sha256":"75829c55e8ef67d02ada77127a5d56bd41806e7fca5e4a9ec60633e4f9b4c130"},"schema_version":"1.0","source":{"id":"2106.06499","kind":"arxiv","version":2}},"canonical_sha256":"da851fe842f7616a92661a1355f2198e9ce3e2a02848c02611542f361f0ea62c","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"da851fe842f7616a92661a1355f2198e9ce3e2a02848c02611542f361f0ea62c","first_computed_at":"2026-07-05T02:51:27.300236Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T02:51:27.300236Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"519L3e4jtpEod9F3MvdVFP7NVIzDFf6bPHg8qKV9fvB1vcJoHI0MBl2/SUFmflgZStJdt89ZH7jVqFfVUHV5DQ==","signature_status":"signed_v1","signed_at":"2026-07-05T02:51:27.300797Z","signed_message":"canonical_sha256_bytes"},"source_id":"2106.06499","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ccc21d46f2d0c69be5c5cd59477516d916bd16b6353fc21a2c264c315f9c7e6a","sha256:bd79a69ce9fc87d212a5bfe2dec0ab80981c1f91dac56aeece01e58b48733bc3"],"state_sha256":"3f7a2b6d157642c4e13f51d82ee22504f899ed8bf85a386b1ac08fe7ddccec4a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0XBhVgjNHir1V514juP8JkuDxTzqzvljH9hqK9mXbD7sWKGB/fIDnULIqWcbHyTiVa+OcxTNgiZfbomntODoDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-03T15:53:27.302007Z","bundle_sha256":"b0ae7ec89d67890e952320298fda683c679d24849d66c42d239f2274804a3b49"}}