{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2018:JZVA6GIIXQRBDQ3A2M4X4IJCZW","short_pith_number":"pith:JZVA6GII","canonical_record":{"source":{"id":"1810.00361","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-09-30T11:29:55Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"46d08747c93dde876cd795186c50916f1a85c865044919ed5be179ddcfd6519f","abstract_canon_sha256":"ebc6f24e8ea05cb5a4645ea818f91d7c2fdf758889baa8603b0eec314b5993e1"},"schema_version":"1.0"},"canonical_sha256":"4e6a0f1908bc2211c360d3397e2122cd9ee09d00c27ce7db6a254a715168f05d","source":{"kind":"arxiv","id":"1810.00361","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1810.00361","created_at":"2026-05-18T00:04:28Z"},{"alias_kind":"arxiv_version","alias_value":"1810.00361v1","created_at":"2026-05-18T00:04:28Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1810.00361","created_at":"2026-05-18T00:04:28Z"},{"alias_kind":"pith_short_12","alias_value":"JZVA6GIIXQRB","created_at":"2026-05-18T12:32:33Z"},{"alias_kind":"pith_short_16","alias_value":"JZVA6GIIXQRBDQ3A","created_at":"2026-05-18T12:32:33Z"},{"alias_kind":"pith_short_8","alias_value":"JZVA6GII","created_at":"2026-05-18T12:32:33Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2018:JZVA6GIIXQRBDQ3A2M4X4IJCZW","target":"record","payload":{"canonical_record":{"source":{"id":"1810.00361","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-09-30T11:29:55Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"46d08747c93dde876cd795186c50916f1a85c865044919ed5be179ddcfd6519f","abstract_canon_sha256":"ebc6f24e8ea05cb5a4645ea818f91d7c2fdf758889baa8603b0eec314b5993e1"},"schema_version":"1.0"},"canonical_sha256":"4e6a0f1908bc2211c360d3397e2122cd9ee09d00c27ce7db6a254a715168f05d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:04:28.066714Z","signature_b64":"UdaQlRtd7zX3O0PYyI+EwlIkOcM1xEzb3OcA1nvJPMS/YUA5P3mE0NI697Mxsu2VFQFSwU2XJw2M6uf25MwyBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4e6a0f1908bc2211c360d3397e2122cd9ee09d00c27ce7db6a254a715168f05d","last_reissued_at":"2026-05-18T00:04:28.066201Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:04:28.066201Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1810.00361","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:04:28Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5fbTbbJYErjRJUqrUuO2SR/0LrNAvgKzLOqQyEAnDgEVjO4XGjRXuAQYseCSVfT/BaVpYE8ER1KsuqZ1Y+EiAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-28T21:22:45.801131Z"},"content_sha256":"9cabce2982e54b721e876a9b17b911a8c50c5b42840dd93849ee8158b39b1823","schema_version":"1.0","event_id":"sha256:9cabce2982e54b721e876a9b17b911a8c50c5b42840dd93849ee8158b39b1823"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2018:JZVA6GIIXQRBDQ3A2M4X4IJCZW","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Using State Predictions for Value Regularization in Curiosity Driven Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Gino Brunner, Manuel Fritsche, Oliver Richter, Roger Wattenhofer","submitted_at":"2018-09-30T11:29:55Z","abstract_excerpt":"Learning in sparse reward settings remains a challenge in Reinforcement Learning, which is often addressed by using intrinsic rewards. One promising strategy is inspired by human curiosity, requiring the agent to learn to predict the future. In this paper a curiosity-driven agent is extended to use these predictions directly for training. To achieve this, the agent predicts the value function of the next state at any point in time. Subsequently, the consistency of this prediction with the current value function is measured, which is then used as a regularization term in the loss function of th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1810.00361","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:04:28Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"m+I2vT/MAMDc6UPdQjUBXuTSlz/oN7AQrSBsMYL7snDyNYWTNnT7OUE1x07kDUAhPGoVGi6yZvfwzW3nSiT0Aw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-28T21:22:45.801815Z"},"content_sha256":"93f57df762baea4a95e69887cb52a868bc4e07cea693d0a44c0297c83839275f","schema_version":"1.0","event_id":"sha256:93f57df762baea4a95e69887cb52a868bc4e07cea693d0a44c0297c83839275f"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/JZVA6GIIXQRBDQ3A2M4X4IJCZW/bundle.json","state_url":"https://pith.science/pith/JZVA6GIIXQRBDQ3A2M4X4IJCZW/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/JZVA6GIIXQRBDQ3A2M4X4IJCZW/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-05-28T21:22:45Z","links":{"resolver":"https://pith.science/pith/JZVA6GIIXQRBDQ3A2M4X4IJCZW","bundle":"https://pith.science/pith/JZVA6GIIXQRBDQ3A2M4X4IJCZW/bundle.json","state":"https://pith.science/pith/JZVA6GIIXQRBDQ3A2M4X4IJCZW/state.json","well_known_bundle":"https://pith.science/.well-known/pith/JZVA6GIIXQRBDQ3A2M4X4IJCZW/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2018:JZVA6GIIXQRBDQ3A2M4X4IJCZW","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ebc6f24e8ea05cb5a4645ea818f91d7c2fdf758889baa8603b0eec314b5993e1","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-09-30T11:29:55Z","title_canon_sha256":"46d08747c93dde876cd795186c50916f1a85c865044919ed5be179ddcfd6519f"},"schema_version":"1.0","source":{"id":"1810.00361","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1810.00361","created_at":"2026-05-18T00:04:28Z"},{"alias_kind":"arxiv_version","alias_value":"1810.00361v1","created_at":"2026-05-18T00:04:28Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1810.00361","created_at":"2026-05-18T00:04:28Z"},{"alias_kind":"pith_short_12","alias_value":"JZVA6GIIXQRB","created_at":"2026-05-18T12:32:33Z"},{"alias_kind":"pith_short_16","alias_value":"JZVA6GIIXQRBDQ3A","created_at":"2026-05-18T12:32:33Z"},{"alias_kind":"pith_short_8","alias_value":"JZVA6GII","created_at":"2026-05-18T12:32:33Z"}],"graph_snapshots":[{"event_id":"sha256:93f57df762baea4a95e69887cb52a868bc4e07cea693d0a44c0297c83839275f","target":"graph","created_at":"2026-05-18T00:04:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"Learning in sparse reward settings remains a challenge in Reinforcement Learning, which is often addressed by using intrinsic rewards. One promising strategy is inspired by human curiosity, requiring the agent to learn to predict the future. In this paper a curiosity-driven agent is extended to use these predictions directly for training. To achieve this, the agent predicts the value function of the next state at any point in time. Subsequently, the consistency of this prediction with the current value function is measured, which is then used as a regularization term in the loss function of th","authors_text":"Gino Brunner, Manuel Fritsche, Oliver Richter, Roger Wattenhofer","cross_cats":["cs.AI","stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-09-30T11:29:55Z","title":"Using State Predictions for Value Regularization in Curiosity Driven Deep Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1810.00361","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:9cabce2982e54b721e876a9b17b911a8c50c5b42840dd93849ee8158b39b1823","target":"record","created_at":"2026-05-18T00:04:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ebc6f24e8ea05cb5a4645ea818f91d7c2fdf758889baa8603b0eec314b5993e1","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-09-30T11:29:55Z","title_canon_sha256":"46d08747c93dde876cd795186c50916f1a85c865044919ed5be179ddcfd6519f"},"schema_version":"1.0","source":{"id":"1810.00361","kind":"arxiv","version":1}},"canonical_sha256":"4e6a0f1908bc2211c360d3397e2122cd9ee09d00c27ce7db6a254a715168f05d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"4e6a0f1908bc2211c360d3397e2122cd9ee09d00c27ce7db6a254a715168f05d","first_computed_at":"2026-05-18T00:04:28.066201Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-18T00:04:28.066201Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"UdaQlRtd7zX3O0PYyI+EwlIkOcM1xEzb3OcA1nvJPMS/YUA5P3mE0NI697Mxsu2VFQFSwU2XJw2M6uf25MwyBQ==","signature_status":"signed_v1","signed_at":"2026-05-18T00:04:28.066714Z","signed_message":"canonical_sha256_bytes"},"source_id":"1810.00361","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:9cabce2982e54b721e876a9b17b911a8c50c5b42840dd93849ee8158b39b1823","sha256:93f57df762baea4a95e69887cb52a868bc4e07cea693d0a44c0297c83839275f"],"state_sha256":"da462ebb8669ff580657cd60fda0cd53c72d51b8bbb610cf7bc55806b989a79e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"TyDFuWqw6s0qHT4mCqKFegTdphPvdykPAYPlGIAky2TnxPoc/k8cfrqq761nIC5ZXrzVzUxt+pav/AGoYgDxCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-05-28T21:22:45.805581Z","bundle_sha256":"322eb853458412022c61a654674581d690c27f85deaf28b5e0f61b99f7ffc765"}}