{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:QDYWMAODHPRVDGVJDEGDX5R74W","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"0c5e5d50726fd03e5af9282039937c6069f4ea3d4386b70a4e0ff13032d97281","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-07T14:03:38Z","title_canon_sha256":"7d5402d5c5d126bf24228c46b3702b2cf6df3f9575e0bac5a3e92aa0d556a8e3"},"schema_version":"1.0","source":{"id":"2002.02794","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2002.02794","created_at":"2026-07-05T00:38:59Z"},{"alias_kind":"arxiv_version","alias_value":"2002.02794v1","created_at":"2026-07-05T00:38:59Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.02794","created_at":"2026-07-05T00:38:59Z"},{"alias_kind":"pith_short_12","alias_value":"QDYWMAODHPRV","created_at":"2026-07-05T00:38:59Z"},{"alias_kind":"pith_short_16","alias_value":"QDYWMAODHPRVDGVJ","created_at":"2026-07-05T00:38:59Z"},{"alias_kind":"pith_short_8","alias_value":"QDYWMAOD","created_at":"2026-07-05T00:38:59Z"}],"graph_snapshots":[{"event_id":"sha256:8fb16147eedb7d51013ffc1134fe8e8c5927faf33df57c6620b70cf13a4f9c32","target":"graph","created_at":"2026-07-05T00:38:59Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2002.02794/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Exploration is widely regarded as one of the most challenging aspects of reinforcement learning (RL), with many naive approaches succumbing to exponential sample complexity. To isolate the challenges of exploration, we propose a new \"reward-free RL\" framework. In the exploration phase, the agent first collects trajectories from an MDP $\\mathcal{M}$ without a pre-specified reward function. After exploration, it is tasked with computing near-optimal policies under for $\\mathcal{M}$ for a collection of given reward functions. This framework is particularly suitable when there are many reward func","authors_text":"Akshay Krishnamurthy, Chi Jin, Max Simchowitz, Tiancheng Yu","cross_cats":["stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-07T14:03:38Z","title":"Reward-Free Exploration for Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.02794","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:fc864c29d0d62c3719587eb4fba54ce35063cada3f5d6fea8bf6e606eb771235","target":"record","created_at":"2026-07-05T00:38:59Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"0c5e5d50726fd03e5af9282039937c6069f4ea3d4386b70a4e0ff13032d97281","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-07T14:03:38Z","title_canon_sha256":"7d5402d5c5d126bf24228c46b3702b2cf6df3f9575e0bac5a3e92aa0d556a8e3"},"schema_version":"1.0","source":{"id":"2002.02794","kind":"arxiv","version":1}},"canonical_sha256":"80f16601c33be3519aa9190c3bf63fe589a84ffbfda9f56ec568f1c5c416606e","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"80f16601c33be3519aa9190c3bf63fe589a84ffbfda9f56ec568f1c5c416606e","first_computed_at":"2026-07-05T00:38:59.366194Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T00:38:59.366194Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"kbLo0xwmrPKpLmX16xITtSLbhP5Agky63cbqB3S/TD3soIxMvgB5vmlIqolFr/Zk3gg1sGXeFjvGtZt6aQnfAg==","signature_status":"signed_v1","signed_at":"2026-07-05T00:38:59.366620Z","signed_message":"canonical_sha256_bytes"},"source_id":"2002.02794","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:fc864c29d0d62c3719587eb4fba54ce35063cada3f5d6fea8bf6e606eb771235","sha256:8fb16147eedb7d51013ffc1134fe8e8c5927faf33df57c6620b70cf13a4f9c32"],"state_sha256":"a8e9638afed9fa3dd0bd4d3149d8a6465f5c358a28b37ec5c0cb3dcfee4c133f"}