{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:IHWSSORF2MKBMSBZC53N5L5JXV","short_pith_number":"pith:IHWSSORF","canonical_record":{"source":{"id":"2101.09207","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-01-22T16:52:06Z","cross_cats_sorted":["cs.IT","math.IT"],"title_canon_sha256":"7e569163247f66999c749c233fc2105581f7e07088fb96c1d439e47e728b63f5","abstract_canon_sha256":"2bedbd6a4c579b01af689b9c55b31e24e4fe29c18ca9def924938d67bd85f3fc"},"schema_version":"1.0"},"canonical_sha256":"41ed293a25d3141648391776deafa9bd4a9431c4c42819ea81d48a258de18b58","source":{"kind":"arxiv","id":"2101.09207","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2101.09207","created_at":"2026-07-05T02:21:19Z"},{"alias_kind":"arxiv_version","alias_value":"2101.09207v2","created_at":"2026-07-05T02:21:19Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.09207","created_at":"2026-07-05T02:21:19Z"},{"alias_kind":"pith_short_12","alias_value":"IHWSSORF2MKB","created_at":"2026-07-05T02:21:19Z"},{"alias_kind":"pith_short_16","alias_value":"IHWSSORF2MKBMSBZ","created_at":"2026-07-05T02:21:19Z"},{"alias_kind":"pith_short_8","alias_value":"IHWSSORF","created_at":"2026-07-05T02:21:19Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:IHWSSORF2MKBMSBZC53N5L5JXV","target":"record","payload":{"canonical_record":{"source":{"id":"2101.09207","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-01-22T16:52:06Z","cross_cats_sorted":["cs.IT","math.IT"],"title_canon_sha256":"7e569163247f66999c749c233fc2105581f7e07088fb96c1d439e47e728b63f5","abstract_canon_sha256":"2bedbd6a4c579b01af689b9c55b31e24e4fe29c18ca9def924938d67bd85f3fc"},"schema_version":"1.0"},"canonical_sha256":"41ed293a25d3141648391776deafa9bd4a9431c4c42819ea81d48a258de18b58","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:21:19.774184Z","signature_b64":"7a/Ga341VGdSDT7QxT+mg4SNbvfZ9RI1ChMz2XP5bcfddhbhqTMzA/lyBaZ3G6aoU9pazn68a5Ns6jYk8LyQAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"41ed293a25d3141648391776deafa9bd4a9431c4c42819ea81d48a258de18b58","last_reissued_at":"2026-07-05T02:21:19.773665Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:21:19.773665Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2101.09207","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:21:19Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"9lnEiWO2FRrCVrC43rsi3XqyEBfpcv1Y9b0JmTVYix0vbCBriKy7mJlwOOFGFShALbjxr2WV7oLeqbdYdKP+AA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T16:23:34.245806Z"},"content_sha256":"3ab0ba360a2c4c14826bc249f799d415cb8cab2fa2b8186a745d51fc2b4d8d05","schema_version":"1.0","event_id":"sha256:3ab0ba360a2c4c14826bc249f799d415cb8cab2fa2b8186a745d51fc2b4d8d05"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:IHWSSORF2MKBMSBZC53N5L5JXV","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Differentiable Trust Region Layers for Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IT","math.IT"],"primary_cat":"cs.LG","authors_text":"Fabian Otto, Gerhard Neumann, Hanna Carolin Ziesche, Ngo Anh Vien, Philipp Becker","submitted_at":"2021-01-22T16:52:06Z","abstract_excerpt":"Trust region methods are a popular tool in reinforcement learning as they yield robust policy updates in continuous and discrete action spaces. However, enforcing such trust regions in deep reinforcement learning is difficult. Hence, many approaches, such as Trust Region Policy Optimization (TRPO) and Proximal Policy Optimization (PPO), are based on approximations. Due to those approximations, they violate the constraints or fail to find the optimal solution within the trust region. Moreover, they are difficult to implement, often lack sufficient exploration, and have been shown to depend on s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.09207","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.09207/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:21:19Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"q9tH7SPULoUPw6Z23K0SAu8YyrKNrOF9Q4P/UGiuiAMojrMnOwEavyQhf/A3ZvBJIck/jmhsaB5OgOYeyh9BDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T16:23:34.246707Z"},"content_sha256":"df8afd7ae173bac722ac4697f466f44501bb7f3eed30a549a14f8f33c9eb286b","schema_version":"1.0","event_id":"sha256:df8afd7ae173bac722ac4697f466f44501bb7f3eed30a549a14f8f33c9eb286b"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/IHWSSORF2MKBMSBZC53N5L5JXV/bundle.json","state_url":"https://pith.science/pith/IHWSSORF2MKBMSBZC53N5L5JXV/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/IHWSSORF2MKBMSBZC53N5L5JXV/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T16:23:34Z","links":{"resolver":"https://pith.science/pith/IHWSSORF2MKBMSBZC53N5L5JXV","bundle":"https://pith.science/pith/IHWSSORF2MKBMSBZC53N5L5JXV/bundle.json","state":"https://pith.science/pith/IHWSSORF2MKBMSBZC53N5L5JXV/state.json","well_known_bundle":"https://pith.science/.well-known/pith/IHWSSORF2MKBMSBZC53N5L5JXV/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:IHWSSORF2MKBMSBZC53N5L5JXV","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"2bedbd6a4c579b01af689b9c55b31e24e4fe29c18ca9def924938d67bd85f3fc","cross_cats_sorted":["cs.IT","math.IT"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-01-22T16:52:06Z","title_canon_sha256":"7e569163247f66999c749c233fc2105581f7e07088fb96c1d439e47e728b63f5"},"schema_version":"1.0","source":{"id":"2101.09207","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2101.09207","created_at":"2026-07-05T02:21:19Z"},{"alias_kind":"arxiv_version","alias_value":"2101.09207v2","created_at":"2026-07-05T02:21:19Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.09207","created_at":"2026-07-05T02:21:19Z"},{"alias_kind":"pith_short_12","alias_value":"IHWSSORF2MKB","created_at":"2026-07-05T02:21:19Z"},{"alias_kind":"pith_short_16","alias_value":"IHWSSORF2MKBMSBZ","created_at":"2026-07-05T02:21:19Z"},{"alias_kind":"pith_short_8","alias_value":"IHWSSORF","created_at":"2026-07-05T02:21:19Z"}],"graph_snapshots":[{"event_id":"sha256:df8afd7ae173bac722ac4697f466f44501bb7f3eed30a549a14f8f33c9eb286b","target":"graph","created_at":"2026-07-05T02:21:19Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2101.09207/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Trust region methods are a popular tool in reinforcement learning as they yield robust policy updates in continuous and discrete action spaces. However, enforcing such trust regions in deep reinforcement learning is difficult. Hence, many approaches, such as Trust Region Policy Optimization (TRPO) and Proximal Policy Optimization (PPO), are based on approximations. Due to those approximations, they violate the constraints or fail to find the optimal solution within the trust region. Moreover, they are difficult to implement, often lack sufficient exploration, and have been shown to depend on s","authors_text":"Fabian Otto, Gerhard Neumann, Hanna Carolin Ziesche, Ngo Anh Vien, Philipp Becker","cross_cats":["cs.IT","math.IT"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-01-22T16:52:06Z","title":"Differentiable Trust Region Layers for Deep Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.09207","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:3ab0ba360a2c4c14826bc249f799d415cb8cab2fa2b8186a745d51fc2b4d8d05","target":"record","created_at":"2026-07-05T02:21:19Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"2bedbd6a4c579b01af689b9c55b31e24e4fe29c18ca9def924938d67bd85f3fc","cross_cats_sorted":["cs.IT","math.IT"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-01-22T16:52:06Z","title_canon_sha256":"7e569163247f66999c749c233fc2105581f7e07088fb96c1d439e47e728b63f5"},"schema_version":"1.0","source":{"id":"2101.09207","kind":"arxiv","version":2}},"canonical_sha256":"41ed293a25d3141648391776deafa9bd4a9431c4c42819ea81d48a258de18b58","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"41ed293a25d3141648391776deafa9bd4a9431c4c42819ea81d48a258de18b58","first_computed_at":"2026-07-05T02:21:19.773665Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T02:21:19.773665Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"7a/Ga341VGdSDT7QxT+mg4SNbvfZ9RI1ChMz2XP5bcfddhbhqTMzA/lyBaZ3G6aoU9pazn68a5Ns6jYk8LyQAA==","signature_status":"signed_v1","signed_at":"2026-07-05T02:21:19.774184Z","signed_message":"canonical_sha256_bytes"},"source_id":"2101.09207","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:3ab0ba360a2c4c14826bc249f799d415cb8cab2fa2b8186a745d51fc2b4d8d05","sha256:df8afd7ae173bac722ac4697f466f44501bb7f3eed30a549a14f8f33c9eb286b"],"state_sha256":"a327e3d353432f4cd4b46e4047bd4ed5b8f1bc0ce9b8c386d989beca33d5c342"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"vkVRxk8tiHaYwMbAmr0jdNaZUvNPHfoRGTVbNK6lxFgDg016NLKpTUFkyOzMhZMwTXriX+gQH5tystk4psNlAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T16:23:34.256851Z","bundle_sha256":"7343fa91ab9e62169435dd9bb189fd9d2ea66575cad20b39ef2f2129359d8341"}}