{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:TAKSR6JGAPUYZCJF7KNMPDKP32","short_pith_number":"pith:TAKSR6JG","schema_version":"1.0","canonical_sha256":"981528f92603e98c8925fa9ac78d4fdeac6cd2f6bb14a5562b9ee9be13d0ad14","source":{"kind":"arxiv","id":"1907.08225","version":4},"attestation_state":"computed","paper":{"title":"Dynamical Distance Learning for Semi-Supervised and Unsupervised Skill Discovery","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Kristian Hartikainen, Sergey Levine, Tuomas Haarnoja, Xinyang Geng","submitted_at":"2019-07-18T18:07:47Z","abstract_excerpt":"Reinforcement learning requires manual specification of a reward function to learn a task. While in principle this reward function only needs to specify the task goal, in practice reinforcement learning can be very time-consuming or even infeasible unless the reward function is shaped so as to provide a smooth gradient towards a successful outcome. This shaping is difficult to specify by hand, particularly when the task is learned from raw observations, such as images. In this paper, we study how we can automatically learn dynamical distances: a measure of the expected number of time steps to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1907.08225","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-07-18T18:07:47Z","cross_cats_sorted":["cs.AI","cs.CV","cs.RO","stat.ML"],"title_canon_sha256":"af0d01b25c18b33e9bef66aba530f2cb552cce8fa0bb7945dd11f975d271dfb2","abstract_canon_sha256":"e385e376b3e09fa010dba34418cdcb7d841cbeb7d846b3c777879c8d423f9bd1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:40:43.554554Z","signature_b64":"43sxZ3yNWQJFw4EYBRJya5eFDQWwwYuzMEvX9nAXURf5A57Gr+41rKGfBQzgdHlVTyGEUgAgvGkAM/4enDGXAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"981528f92603e98c8925fa9ac78d4fdeac6cd2f6bb14a5562b9ee9be13d0ad14","last_reissued_at":"2026-07-05T00:40:43.554003Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:40:43.554003Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dynamical Distance Learning for Semi-Supervised and Unsupervised Skill Discovery","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Kristian Hartikainen, Sergey Levine, Tuomas Haarnoja, Xinyang Geng","submitted_at":"2019-07-18T18:07:47Z","abstract_excerpt":"Reinforcement learning requires manual specification of a reward function to learn a task. While in principle this reward function only needs to specify the task goal, in practice reinforcement learning can be very time-consuming or even infeasible unless the reward function is shaped so as to provide a smooth gradient towards a successful outcome. This shaping is difficult to specify by hand, particularly when the task is learned from raw observations, such as images. In this paper, we study how we can automatically learn dynamical distances: a measure of the expected number of time steps to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1907.08225","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1907.08225/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1907.08225","created_at":"2026-07-05T00:40:43.554085+00:00"},{"alias_kind":"arxiv_version","alias_value":"1907.08225v4","created_at":"2026-07-05T00:40:43.554085+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1907.08225","created_at":"2026-07-05T00:40:43.554085+00:00"},{"alias_kind":"pith_short_12","alias_value":"TAKSR6JGAPUY","created_at":"2026-07-05T00:40:43.554085+00:00"},{"alias_kind":"pith_short_16","alias_value":"TAKSR6JGAPUYZCJF","created_at":"2026-07-05T00:40:43.554085+00:00"},{"alias_kind":"pith_short_8","alias_value":"TAKSR6JG","created_at":"2026-07-05T00:40:43.554085+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.01862","citing_title":"QHyer: Q-conditioned Hybrid Attention-mamba Transformer for Offline Goal-conditioned RL","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20627","citing_title":"Occupancy Reward Shaping: Improving Credit Assignment for Offline Goal-Conditioned Reinforcement Learning","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TAKSR6JGAPUYZCJF7KNMPDKP32","json":"https://pith.science/pith/TAKSR6JGAPUYZCJF7KNMPDKP32.json","graph_json":"https://pith.science/api/pith-number/TAKSR6JGAPUYZCJF7KNMPDKP32/graph.json","events_json":"https://pith.science/api/pith-number/TAKSR6JGAPUYZCJF7KNMPDKP32/events.json","paper":"https://pith.science/paper/TAKSR6JG"},"agent_actions":{"view_html":"https://pith.science/pith/TAKSR6JGAPUYZCJF7KNMPDKP32","download_json":"https://pith.science/pith/TAKSR6JGAPUYZCJF7KNMPDKP32.json","view_paper":"https://pith.science/paper/TAKSR6JG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1907.08225&json=true","fetch_graph":"https://pith.science/api/pith-number/TAKSR6JGAPUYZCJF7KNMPDKP32/graph.json","fetch_events":"https://pith.science/api/pith-number/TAKSR6JGAPUYZCJF7KNMPDKP32/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TAKSR6JGAPUYZCJF7KNMPDKP32/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TAKSR6JGAPUYZCJF7KNMPDKP32/action/storage_attestation","attest_author":"https://pith.science/pith/TAKSR6JGAPUYZCJF7KNMPDKP32/action/author_attestation","sign_citation":"https://pith.science/pith/TAKSR6JGAPUYZCJF7KNMPDKP32/action/citation_signature","submit_replication":"https://pith.science/pith/TAKSR6JGAPUYZCJF7KNMPDKP32/action/replication_record"}},"created_at":"2026-07-05T00:40:43.554085+00:00","updated_at":"2026-07-05T00:40:43.554085+00:00"}