{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:2XMX3DZDIGDCHKQQIPQUL7ZNA6","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"24970595b2da5d1176423e500a8010babe370a63f14c3ec306e29b1637cf27f0","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2020-04-27T17:38:53Z","title_canon_sha256":"83186ac11974d035831b402b29210e6703c23713dec816855828f37268388003"},"schema_version":"1.0","source":{"id":"2004.12974","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2004.12974","created_at":"2026-07-05T00:58:29Z"},{"alias_kind":"arxiv_version","alias_value":"2004.12974v1","created_at":"2026-07-05T00:58:29Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.12974","created_at":"2026-07-05T00:58:29Z"},{"alias_kind":"pith_short_12","alias_value":"2XMX3DZDIGDC","created_at":"2026-07-05T00:58:29Z"},{"alias_kind":"pith_short_16","alias_value":"2XMX3DZDIGDCHKQQ","created_at":"2026-07-05T00:58:29Z"},{"alias_kind":"pith_short_8","alias_value":"2XMX3DZD","created_at":"2026-07-05T00:58:29Z"}],"graph_snapshots":[{"event_id":"sha256:89ec3f5f2832d42fe222b7c628549cffdaaeecf1e6c1e81414970e2e86a0dfd8","target":"graph","created_at":"2026-07-05T00:58:29Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2004.12974/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning provides a general framework for learning robotic skills while minimizing engineering effort. However, most reinforcement learning algorithms assume that a well-designed reward function is provided, and learn a single behavior for that single reward function. Such reward functions can be difficult to design in practice. Can we instead develop efficient reinforcement learning methods that acquire diverse skills without any reward function, and then repurpose these skills for downstream tasks? In this paper, we demonstrate that a recently proposed unsupervised skill discov","authors_text":"Archit Sharma, Karol Hausman, Michael Ahn, Sergey Levine, Shixiang Gu, Vikash Kumar","cross_cats":["cs.LG"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2020-04-27T17:38:53Z","title":"Emergent Real-World Robotic Skills via Unsupervised Off-Policy Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.12974","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:50e595f3110162de149886b16fa45e6eca6be477932389741d99178816ab2228","target":"record","created_at":"2026-07-05T00:58:29Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"24970595b2da5d1176423e500a8010babe370a63f14c3ec306e29b1637cf27f0","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2020-04-27T17:38:53Z","title_canon_sha256":"83186ac11974d035831b402b29210e6703c23713dec816855828f37268388003"},"schema_version":"1.0","source":{"id":"2004.12974","kind":"arxiv","version":1}},"canonical_sha256":"d5d97d8f23418623aa1043e145ff2d0782ea3f7c3f2e784be529f5e590c3314f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d5d97d8f23418623aa1043e145ff2d0782ea3f7c3f2e784be529f5e590c3314f","first_computed_at":"2026-07-05T00:58:29.176207Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T00:58:29.176207Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"MWsnx7aTzxg5sq2mJF9UsPyrlZQ6RnGwUlbzvsaiHUPXHTVRhbKg0htiLyshsxPyZDC9u3xsnk78j0C73eOtDg==","signature_status":"signed_v1","signed_at":"2026-07-05T00:58:29.176668Z","signed_message":"canonical_sha256_bytes"},"source_id":"2004.12974","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:50e595f3110162de149886b16fa45e6eca6be477932389741d99178816ab2228","sha256:89ec3f5f2832d42fe222b7c628549cffdaaeecf1e6c1e81414970e2e86a0dfd8"],"state_sha256":"918a0fd823dcdb9892a135c06cf6f7dc5efc84ec797babaed6b753b613eff0e0"}