{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:BX53AGQ23IROHG7RKSHGPFRQJ2","short_pith_number":"pith:BX53AGQ2","canonical_record":{"source":{"id":"2406.16255","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-24T01:37:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"efc92011411bbd93c2f91175a8939e34634ba47d8180c46296b958b320537234","abstract_canon_sha256":"f70d36328937362d9d34617f805cd276f4d73c1c36274d3b22ced303c62e2204"},"schema_version":"1.0"},"canonical_sha256":"0dfbb01a1ada22e39bf1548e6796304e8684b720e4a936d29a325f2539baf24d","source":{"kind":"arxiv","id":"2406.16255","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.16255","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"arxiv_version","alias_value":"2406.16255v2","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.16255","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"pith_short_12","alias_value":"BX53AGQ23IRO","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"pith_short_16","alias_value":"BX53AGQ23IROHG7R","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"pith_short_8","alias_value":"BX53AGQ2","created_at":"2026-07-05T08:38:11Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:BX53AGQ23IROHG7RKSHGPFRQJ2","target":"record","payload":{"canonical_record":{"source":{"id":"2406.16255","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-24T01:37:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"efc92011411bbd93c2f91175a8939e34634ba47d8180c46296b958b320537234","abstract_canon_sha256":"f70d36328937362d9d34617f805cd276f4d73c1c36274d3b22ced303c62e2204"},"schema_version":"1.0"},"canonical_sha256":"0dfbb01a1ada22e39bf1548e6796304e8684b720e4a936d29a325f2539baf24d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:38:11.705610Z","signature_b64":"oZdpc9/d9ejRCddMEhtqj6KxBMRyVGMLzEZiEm48JZdWURZRNaGlQpyGDTFIH69Tlp8yZpTn6Q0wMH0LX/ziCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0dfbb01a1ada22e39bf1548e6796304e8684b720e4a936d29a325f2539baf24d","last_reissued_at":"2026-07-05T08:38:11.705122Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:38:11.705122Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2406.16255","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:38:11Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"JIHkNqybVabPPAZy0gJMVdnNNvwTPTm37nBIkJB47oIEHhN6lWGp7xAgoPp+vcaVCB7NP/ZOMZVhaLjcMI8RDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T07:51:08.763699Z"},"content_sha256":"408043f933ccf49d86d7cc8bc938821ad4014444beea340b118dbf64ce316475","schema_version":"1.0","event_id":"sha256:408043f933ccf49d86d7cc8bc938821ad4014444beea340b118dbf64ce316475"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:BX53AGQ23IROHG7RKSHGPFRQJ2","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Uncertainty-Aware Reward-Free Exploration with General Function Approximation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dongruo Zhou, Junkai Zhang, Quanquan Gu, Weitong Zhang","submitted_at":"2024-06-24T01:37:18Z","abstract_excerpt":"Mastering multiple tasks through exploration and learning in an environment poses a significant challenge in reinforcement learning (RL). Unsupervised RL has been introduced to address this challenge by training policies with intrinsic rewards rather than extrinsic rewards. However, current intrinsic reward designs and unsupervised RL algorithms often overlook the heterogeneous nature of collected samples, thereby diminishing their sample efficiency. To overcome this limitation, in this paper, we propose a reward-free RL algorithm called \\alg. The key idea behind our algorithm is an uncertaint"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.16255","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.16255/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:38:11Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"fq5PJfrbd7mLut6qOiQCPwC1WtpNpUOMe19l8W2vkeHekO+i/sjeBS6+2fmiNHDGX+PnrWC8AW4uY6DgiHI5CA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T07:51:08.764232Z"},"content_sha256":"d3d431b52c22af47a23d78cdb24d6881b499062334ecb01b35239577b87f0aa4","schema_version":"1.0","event_id":"sha256:d3d431b52c22af47a23d78cdb24d6881b499062334ecb01b35239577b87f0aa4"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/BX53AGQ23IROHG7RKSHGPFRQJ2/bundle.json","state_url":"https://pith.science/pith/BX53AGQ23IROHG7RKSHGPFRQJ2/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/BX53AGQ23IROHG7RKSHGPFRQJ2/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-18T07:51:08Z","links":{"resolver":"https://pith.science/pith/BX53AGQ23IROHG7RKSHGPFRQJ2","bundle":"https://pith.science/pith/BX53AGQ23IROHG7RKSHGPFRQJ2/bundle.json","state":"https://pith.science/pith/BX53AGQ23IROHG7RKSHGPFRQJ2/state.json","well_known_bundle":"https://pith.science/.well-known/pith/BX53AGQ23IROHG7RKSHGPFRQJ2/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:BX53AGQ23IROHG7RKSHGPFRQJ2","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f70d36328937362d9d34617f805cd276f4d73c1c36274d3b22ced303c62e2204","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-24T01:37:18Z","title_canon_sha256":"efc92011411bbd93c2f91175a8939e34634ba47d8180c46296b958b320537234"},"schema_version":"1.0","source":{"id":"2406.16255","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.16255","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"arxiv_version","alias_value":"2406.16255v2","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.16255","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"pith_short_12","alias_value":"BX53AGQ23IRO","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"pith_short_16","alias_value":"BX53AGQ23IROHG7R","created_at":"2026-07-05T08:38:11Z"},{"alias_kind":"pith_short_8","alias_value":"BX53AGQ2","created_at":"2026-07-05T08:38:11Z"}],"graph_snapshots":[{"event_id":"sha256:d3d431b52c22af47a23d78cdb24d6881b499062334ecb01b35239577b87f0aa4","target":"graph","created_at":"2026-07-05T08:38:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2406.16255/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Mastering multiple tasks through exploration and learning in an environment poses a significant challenge in reinforcement learning (RL). Unsupervised RL has been introduced to address this challenge by training policies with intrinsic rewards rather than extrinsic rewards. However, current intrinsic reward designs and unsupervised RL algorithms often overlook the heterogeneous nature of collected samples, thereby diminishing their sample efficiency. To overcome this limitation, in this paper, we propose a reward-free RL algorithm called \\alg. The key idea behind our algorithm is an uncertaint","authors_text":"Dongruo Zhou, Junkai Zhang, Quanquan Gu, Weitong Zhang","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-24T01:37:18Z","title":"Uncertainty-Aware Reward-Free Exploration with General Function Approximation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.16255","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:408043f933ccf49d86d7cc8bc938821ad4014444beea340b118dbf64ce316475","target":"record","created_at":"2026-07-05T08:38:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f70d36328937362d9d34617f805cd276f4d73c1c36274d3b22ced303c62e2204","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-24T01:37:18Z","title_canon_sha256":"efc92011411bbd93c2f91175a8939e34634ba47d8180c46296b958b320537234"},"schema_version":"1.0","source":{"id":"2406.16255","kind":"arxiv","version":2}},"canonical_sha256":"0dfbb01a1ada22e39bf1548e6796304e8684b720e4a936d29a325f2539baf24d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"0dfbb01a1ada22e39bf1548e6796304e8684b720e4a936d29a325f2539baf24d","first_computed_at":"2026-07-05T08:38:11.705122Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:38:11.705122Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"oZdpc9/d9ejRCddMEhtqj6KxBMRyVGMLzEZiEm48JZdWURZRNaGlQpyGDTFIH69Tlp8yZpTn6Q0wMH0LX/ziCA==","signature_status":"signed_v1","signed_at":"2026-07-05T08:38:11.705610Z","signed_message":"canonical_sha256_bytes"},"source_id":"2406.16255","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:408043f933ccf49d86d7cc8bc938821ad4014444beea340b118dbf64ce316475","sha256:d3d431b52c22af47a23d78cdb24d6881b499062334ecb01b35239577b87f0aa4"],"state_sha256":"f25bc0e69c52ad60229d9b32d9a1f16e740aeb3ad6214f801cf0b67a49d352cb"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ZOa/NB5psr59Een1KDEyw4KYJWgHzvIBZMzBjWTApxqPKjdqL0dRrCdvT81LX6XNXm5yipYPCPDodemTvtROAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-18T07:51:08.769798Z","bundle_sha256":"676ebc88e0f7d5d8d1e3dac81180b43deaf0e8c4cd0f5204842d7b7185d03db5"}}