{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:22UV64IJG437G3VBHUPOJTKHRF","short_pith_number":"pith:22UV64IJ","canonical_record":{"source":{"id":"2210.06702","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-13T03:42:19Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"da67069ad716df0ac4f9cd90fec6341345c301e0ff40b7716c689cf8ace05c6e","abstract_canon_sha256":"27ea1db1da2c7f076e43bf3860a39863aaa8e861964c29da57cf185191a18432"},"schema_version":"1.0"},"canonical_sha256":"d6a95f71093737f36ea13d1ee4cd47896b4b38aa35571462547b92f889f56054","source":{"kind":"arxiv","id":"2210.06702","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2210.06702","created_at":"2026-07-05T05:06:18Z"},{"alias_kind":"arxiv_version","alias_value":"2210.06702v1","created_at":"2026-07-05T05:06:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.06702","created_at":"2026-07-05T05:06:18Z"},{"alias_kind":"pith_short_12","alias_value":"22UV64IJG437","created_at":"2026-07-05T05:06:18Z"},{"alias_kind":"pith_short_16","alias_value":"22UV64IJG437G3VB","created_at":"2026-07-05T05:06:18Z"},{"alias_kind":"pith_short_8","alias_value":"22UV64IJ","created_at":"2026-07-05T05:06:18Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:22UV64IJG437G3VBHUPOJTKHRF","target":"record","payload":{"canonical_record":{"source":{"id":"2210.06702","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-13T03:42:19Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"da67069ad716df0ac4f9cd90fec6341345c301e0ff40b7716c689cf8ace05c6e","abstract_canon_sha256":"27ea1db1da2c7f076e43bf3860a39863aaa8e861964c29da57cf185191a18432"},"schema_version":"1.0"},"canonical_sha256":"d6a95f71093737f36ea13d1ee4cd47896b4b38aa35571462547b92f889f56054","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:06:18.945181Z","signature_b64":"IKbFccBn3OCbNyEWRkWTMehgzYaP/AmrtfiEdYT2AynbGDHKKoYdSd/Q45JpFPDa5alVpixrOAgkvomT1al0AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d6a95f71093737f36ea13d1ee4cd47896b4b38aa35571462547b92f889f56054","last_reissued_at":"2026-07-05T05:06:18.944650Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:06:18.944650Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2210.06702","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:06:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ZbQOqZaqnq4RSGAyLCUz56+43fIP/2Fv4JfZuQ8ms5DcWC1Ib52FpRzATtWqI5ZDS80/Fi8srp0IaO7pHH2ZCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-21T12:39:40.690499Z"},"content_sha256":"08d11e42d02f3050925f5690c4a2e0aedeb8bed3cb6b105229c2a8b4099e268b","schema_version":"1.0","event_id":"sha256:08d11e42d02f3050925f5690c4a2e0aedeb8bed3cb6b105229c2a8b4099e268b"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:22UV64IJG437G3VBHUPOJTKHRF","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"A Mixture of Surprises for Unsupervised Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Andrew Zhao, Gao Huang, Matthieu Gaetan Lin, Yangguang Li, Yong-jin Liu","submitted_at":"2022-10-13T03:42:19Z","abstract_excerpt":"Unsupervised reinforcement learning aims at learning a generalist policy in a reward-free manner for fast adaptation to downstream tasks. Most of the existing methods propose to provide an intrinsic reward based on surprise. Maximizing or minimizing surprise drives the agent to either explore or gain control over its environment. However, both strategies rely on a strong assumption: the entropy of the environment's dynamics is either high or low. This assumption may not always hold in real-world scenarios, where the entropy of the environment's dynamics may be unknown. Hence, choosing between "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.06702","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.06702/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:06:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"265OJAoLmZxgXXddbCioXkUyN2QqUhi2SBTfKdvVjb9EkVHh5lwF149wf4Qxcc7WRyKJYeHas53G7WaT8MyhCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-21T12:39:40.690870Z"},"content_sha256":"6ecdc39e0f5723d3ded09aed52b3e662e888f8a8913da1955dc7191db8948858","schema_version":"1.0","event_id":"sha256:6ecdc39e0f5723d3ded09aed52b3e662e888f8a8913da1955dc7191db8948858"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/22UV64IJG437G3VBHUPOJTKHRF/bundle.json","state_url":"https://pith.science/pith/22UV64IJG437G3VBHUPOJTKHRF/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/22UV64IJG437G3VBHUPOJTKHRF/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-21T12:39:40Z","links":{"resolver":"https://pith.science/pith/22UV64IJG437G3VBHUPOJTKHRF","bundle":"https://pith.science/pith/22UV64IJG437G3VBHUPOJTKHRF/bundle.json","state":"https://pith.science/pith/22UV64IJG437G3VBHUPOJTKHRF/state.json","well_known_bundle":"https://pith.science/.well-known/pith/22UV64IJG437G3VBHUPOJTKHRF/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:22UV64IJG437G3VBHUPOJTKHRF","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"27ea1db1da2c7f076e43bf3860a39863aaa8e861964c29da57cf185191a18432","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-13T03:42:19Z","title_canon_sha256":"da67069ad716df0ac4f9cd90fec6341345c301e0ff40b7716c689cf8ace05c6e"},"schema_version":"1.0","source":{"id":"2210.06702","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2210.06702","created_at":"2026-07-05T05:06:18Z"},{"alias_kind":"arxiv_version","alias_value":"2210.06702v1","created_at":"2026-07-05T05:06:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.06702","created_at":"2026-07-05T05:06:18Z"},{"alias_kind":"pith_short_12","alias_value":"22UV64IJG437","created_at":"2026-07-05T05:06:18Z"},{"alias_kind":"pith_short_16","alias_value":"22UV64IJG437G3VB","created_at":"2026-07-05T05:06:18Z"},{"alias_kind":"pith_short_8","alias_value":"22UV64IJ","created_at":"2026-07-05T05:06:18Z"}],"graph_snapshots":[{"event_id":"sha256:6ecdc39e0f5723d3ded09aed52b3e662e888f8a8913da1955dc7191db8948858","target":"graph","created_at":"2026-07-05T05:06:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2210.06702/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Unsupervised reinforcement learning aims at learning a generalist policy in a reward-free manner for fast adaptation to downstream tasks. Most of the existing methods propose to provide an intrinsic reward based on surprise. Maximizing or minimizing surprise drives the agent to either explore or gain control over its environment. However, both strategies rely on a strong assumption: the entropy of the environment's dynamics is either high or low. This assumption may not always hold in real-world scenarios, where the entropy of the environment's dynamics may be unknown. Hence, choosing between ","authors_text":"Andrew Zhao, Gao Huang, Matthieu Gaetan Lin, Yangguang Li, Yong-jin Liu","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-13T03:42:19Z","title":"A Mixture of Surprises for Unsupervised Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.06702","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:08d11e42d02f3050925f5690c4a2e0aedeb8bed3cb6b105229c2a8b4099e268b","target":"record","created_at":"2026-07-05T05:06:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"27ea1db1da2c7f076e43bf3860a39863aaa8e861964c29da57cf185191a18432","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-13T03:42:19Z","title_canon_sha256":"da67069ad716df0ac4f9cd90fec6341345c301e0ff40b7716c689cf8ace05c6e"},"schema_version":"1.0","source":{"id":"2210.06702","kind":"arxiv","version":1}},"canonical_sha256":"d6a95f71093737f36ea13d1ee4cd47896b4b38aa35571462547b92f889f56054","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d6a95f71093737f36ea13d1ee4cd47896b4b38aa35571462547b92f889f56054","first_computed_at":"2026-07-05T05:06:18.944650Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:06:18.944650Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"IKbFccBn3OCbNyEWRkWTMehgzYaP/AmrtfiEdYT2AynbGDHKKoYdSd/Q45JpFPDa5alVpixrOAgkvomT1al0AQ==","signature_status":"signed_v1","signed_at":"2026-07-05T05:06:18.945181Z","signed_message":"canonical_sha256_bytes"},"source_id":"2210.06702","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:08d11e42d02f3050925f5690c4a2e0aedeb8bed3cb6b105229c2a8b4099e268b","sha256:6ecdc39e0f5723d3ded09aed52b3e662e888f8a8913da1955dc7191db8948858"],"state_sha256":"903dcbde3a21688a2e2a9f9ba6834a5a67dbb85264f7c8c0fc39f2a721e642f0"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"HA1J7oS2Kbwtf6RF3rO4Ydnl5wKlhtJnmWEFrTOpHrnHMIpgyZoLeo2LaMBtGTaKIkl0AGSFqs4+GlCXSoLLCQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-21T12:39:40.693030Z","bundle_sha256":"6d6d664d1cf5fee24908f0b0e490b409c4c4cda9d2fdf513443ebcd680eacb63"}}