{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2019:TAKSR6JGAPUYZCJF7KNMPDKP32","short_pith_number":"pith:TAKSR6JG","canonical_record":{"source":{"id":"1907.08225","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-07-18T18:07:47Z","cross_cats_sorted":["cs.AI","cs.CV","cs.RO","stat.ML"],"title_canon_sha256":"af0d01b25c18b33e9bef66aba530f2cb552cce8fa0bb7945dd11f975d271dfb2","abstract_canon_sha256":"e385e376b3e09fa010dba34418cdcb7d841cbeb7d846b3c777879c8d423f9bd1"},"schema_version":"1.0"},"canonical_sha256":"981528f92603e98c8925fa9ac78d4fdeac6cd2f6bb14a5562b9ee9be13d0ad14","source":{"kind":"arxiv","id":"1907.08225","version":4},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1907.08225","created_at":"2026-07-05T00:40:43Z"},{"alias_kind":"arxiv_version","alias_value":"1907.08225v4","created_at":"2026-07-05T00:40:43Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1907.08225","created_at":"2026-07-05T00:40:43Z"},{"alias_kind":"pith_short_12","alias_value":"TAKSR6JGAPUY","created_at":"2026-07-05T00:40:43Z"},{"alias_kind":"pith_short_16","alias_value":"TAKSR6JGAPUYZCJF","created_at":"2026-07-05T00:40:43Z"},{"alias_kind":"pith_short_8","alias_value":"TAKSR6JG","created_at":"2026-07-05T00:40:43Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2019:TAKSR6JGAPUYZCJF7KNMPDKP32","target":"record","payload":{"canonical_record":{"source":{"id":"1907.08225","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-07-18T18:07:47Z","cross_cats_sorted":["cs.AI","cs.CV","cs.RO","stat.ML"],"title_canon_sha256":"af0d01b25c18b33e9bef66aba530f2cb552cce8fa0bb7945dd11f975d271dfb2","abstract_canon_sha256":"e385e376b3e09fa010dba34418cdcb7d841cbeb7d846b3c777879c8d423f9bd1"},"schema_version":"1.0"},"canonical_sha256":"981528f92603e98c8925fa9ac78d4fdeac6cd2f6bb14a5562b9ee9be13d0ad14","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:40:43.554554Z","signature_b64":"43sxZ3yNWQJFw4EYBRJya5eFDQWwwYuzMEvX9nAXURf5A57Gr+41rKGfBQzgdHlVTyGEUgAgvGkAM/4enDGXAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"981528f92603e98c8925fa9ac78d4fdeac6cd2f6bb14a5562b9ee9be13d0ad14","last_reissued_at":"2026-07-05T00:40:43.554003Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:40:43.554003Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1907.08225","source_version":4,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T00:40:43Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"bfXdJ+aF6WWmdTwMzeHmRwcHJOEd0ZZikMPddfQwd3sLa2CBs2FDQdtC1XYRURYiykgjVjKrT7LgJqBs8zduAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T20:57:15.850987Z"},"content_sha256":"11f04326a932bd9cb3b538310801310ba66fa7d7fe377bd0960a03ef7ae9f7da","schema_version":"1.0","event_id":"sha256:11f04326a932bd9cb3b538310801310ba66fa7d7fe377bd0960a03ef7ae9f7da"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2019:TAKSR6JGAPUYZCJF7KNMPDKP32","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Dynamical Distance Learning for Semi-Supervised and Unsupervised Skill Discovery","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Kristian Hartikainen, Sergey Levine, Tuomas Haarnoja, Xinyang Geng","submitted_at":"2019-07-18T18:07:47Z","abstract_excerpt":"Reinforcement learning requires manual specification of a reward function to learn a task. While in principle this reward function only needs to specify the task goal, in practice reinforcement learning can be very time-consuming or even infeasible unless the reward function is shaped so as to provide a smooth gradient towards a successful outcome. This shaping is difficult to specify by hand, particularly when the task is learned from raw observations, such as images. In this paper, we study how we can automatically learn dynamical distances: a measure of the expected number of time steps to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1907.08225","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1907.08225/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T00:40:43Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"X5p/QPd/ADrzWOusXIJJa/2RYsZwp+DRfRD9tr3FXkolcBmlUYeuBtTn990hiBrqRvoB9sHjxvTg2U09c34rCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T20:57:15.851494Z"},"content_sha256":"e09b8dc6214cfa0b28d12802dd98e0f135093b7842c3e341a3a8c17d4e38afe8","schema_version":"1.0","event_id":"sha256:e09b8dc6214cfa0b28d12802dd98e0f135093b7842c3e341a3a8c17d4e38afe8"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/TAKSR6JGAPUYZCJF7KNMPDKP32/bundle.json","state_url":"https://pith.science/pith/TAKSR6JGAPUYZCJF7KNMPDKP32/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/TAKSR6JGAPUYZCJF7KNMPDKP32/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T20:57:15Z","links":{"resolver":"https://pith.science/pith/TAKSR6JGAPUYZCJF7KNMPDKP32","bundle":"https://pith.science/pith/TAKSR6JGAPUYZCJF7KNMPDKP32/bundle.json","state":"https://pith.science/pith/TAKSR6JGAPUYZCJF7KNMPDKP32/state.json","well_known_bundle":"https://pith.science/.well-known/pith/TAKSR6JGAPUYZCJF7KNMPDKP32/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2019:TAKSR6JGAPUYZCJF7KNMPDKP32","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"e385e376b3e09fa010dba34418cdcb7d841cbeb7d846b3c777879c8d423f9bd1","cross_cats_sorted":["cs.AI","cs.CV","cs.RO","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-07-18T18:07:47Z","title_canon_sha256":"af0d01b25c18b33e9bef66aba530f2cb552cce8fa0bb7945dd11f975d271dfb2"},"schema_version":"1.0","source":{"id":"1907.08225","kind":"arxiv","version":4}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1907.08225","created_at":"2026-07-05T00:40:43Z"},{"alias_kind":"arxiv_version","alias_value":"1907.08225v4","created_at":"2026-07-05T00:40:43Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1907.08225","created_at":"2026-07-05T00:40:43Z"},{"alias_kind":"pith_short_12","alias_value":"TAKSR6JGAPUY","created_at":"2026-07-05T00:40:43Z"},{"alias_kind":"pith_short_16","alias_value":"TAKSR6JGAPUYZCJF","created_at":"2026-07-05T00:40:43Z"},{"alias_kind":"pith_short_8","alias_value":"TAKSR6JG","created_at":"2026-07-05T00:40:43Z"}],"graph_snapshots":[{"event_id":"sha256:e09b8dc6214cfa0b28d12802dd98e0f135093b7842c3e341a3a8c17d4e38afe8","target":"graph","created_at":"2026-07-05T00:40:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/1907.08225/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning requires manual specification of a reward function to learn a task. While in principle this reward function only needs to specify the task goal, in practice reinforcement learning can be very time-consuming or even infeasible unless the reward function is shaped so as to provide a smooth gradient towards a successful outcome. This shaping is difficult to specify by hand, particularly when the task is learned from raw observations, such as images. In this paper, we study how we can automatically learn dynamical distances: a measure of the expected number of time steps to ","authors_text":"Kristian Hartikainen, Sergey Levine, Tuomas Haarnoja, Xinyang Geng","cross_cats":["cs.AI","cs.CV","cs.RO","stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-07-18T18:07:47Z","title":"Dynamical Distance Learning for Semi-Supervised and Unsupervised Skill Discovery"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1907.08225","kind":"arxiv","version":4},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:11f04326a932bd9cb3b538310801310ba66fa7d7fe377bd0960a03ef7ae9f7da","target":"record","created_at":"2026-07-05T00:40:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"e385e376b3e09fa010dba34418cdcb7d841cbeb7d846b3c777879c8d423f9bd1","cross_cats_sorted":["cs.AI","cs.CV","cs.RO","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-07-18T18:07:47Z","title_canon_sha256":"af0d01b25c18b33e9bef66aba530f2cb552cce8fa0bb7945dd11f975d271dfb2"},"schema_version":"1.0","source":{"id":"1907.08225","kind":"arxiv","version":4}},"canonical_sha256":"981528f92603e98c8925fa9ac78d4fdeac6cd2f6bb14a5562b9ee9be13d0ad14","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"981528f92603e98c8925fa9ac78d4fdeac6cd2f6bb14a5562b9ee9be13d0ad14","first_computed_at":"2026-07-05T00:40:43.554003Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T00:40:43.554003Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"43sxZ3yNWQJFw4EYBRJya5eFDQWwwYuzMEvX9nAXURf5A57Gr+41rKGfBQzgdHlVTyGEUgAgvGkAM/4enDGXAw==","signature_status":"signed_v1","signed_at":"2026-07-05T00:40:43.554554Z","signed_message":"canonical_sha256_bytes"},"source_id":"1907.08225","source_kind":"arxiv","source_version":4}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:11f04326a932bd9cb3b538310801310ba66fa7d7fe377bd0960a03ef7ae9f7da","sha256:e09b8dc6214cfa0b28d12802dd98e0f135093b7842c3e341a3a8c17d4e38afe8"],"state_sha256":"f2c00db65898874b4acc50df20576438a966cff5ff8eb0d1c8b297a2058a21e2"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"xMlANi6eviYP80evuqMzKt4q6pMNSWcJVAX/hglTEGLF188XwE20Ev9IXYiIH/y1dPUnmVF3oLS6xNkV2AqPBQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T20:57:15.855422Z","bundle_sha256":"7de8ebf6f680f61145e312c892fae027d4da09fe98360b1e1dc07d1856e0979a"}}