{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:EDQMQJT73JC5DHML4KTRXNBVKK","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d6ef1c48a8b28c6981066b4d53ed6b2217821ac66288fe2f3148600df5b77d3c","cross_cats_sorted":["cs.AI","math.OC","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-06-19T17:59:36Z","title_canon_sha256":"10f836635f454fbbbdac8f0a867833bb462203215079cbdfa3f6e9fc4114ad90"},"schema_version":"1.0","source":{"id":"2006.11274","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2006.11274","created_at":"2026-07-05T01:11:37Z"},{"alias_kind":"arxiv_version","alias_value":"2006.11274v1","created_at":"2026-07-05T01:11:37Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.11274","created_at":"2026-07-05T01:11:37Z"},{"alias_kind":"pith_short_12","alias_value":"EDQMQJT73JC5","created_at":"2026-07-05T01:11:37Z"},{"alias_kind":"pith_short_16","alias_value":"EDQMQJT73JC5DHML","created_at":"2026-07-05T01:11:37Z"},{"alias_kind":"pith_short_8","alias_value":"EDQMQJT7","created_at":"2026-07-05T01:11:37Z"}],"graph_snapshots":[{"event_id":"sha256:e67a963ca837b1f59f219327a190ba82a13c140938727f40b3a5cc837e4c86a8","target":"graph","created_at":"2026-07-05T01:11:37Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2006.11274/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reward-free reinforcement learning (RL) is a framework which is suitable for both the batch RL setting and the setting where there are many reward functions of interest. During the exploration phase, an agent collects samples without using a pre-specified reward function. After the exploration phase, a reward function is given, and the agent uses samples collected during the exploration phase to compute a near-optimal policy. Jin et al. [2020] showed that in the tabular setting, the agent only needs to collect polynomial number of samples (in terms of the number states, the number of actions, ","authors_text":"Lin F. Yang, Ruosong Wang, Ruslan Salakhutdinov, Simon S. Du","cross_cats":["cs.AI","math.OC","stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-06-19T17:59:36Z","title":"On Reward-Free Reinforcement Learning with Linear Function Approximation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.11274","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:1fe12b99a426f4e189bead445cbaf7243637de2e90ecb49d5317fb46f7a7aaa3","target":"record","created_at":"2026-07-05T01:11:37Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d6ef1c48a8b28c6981066b4d53ed6b2217821ac66288fe2f3148600df5b77d3c","cross_cats_sorted":["cs.AI","math.OC","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-06-19T17:59:36Z","title_canon_sha256":"10f836635f454fbbbdac8f0a867833bb462203215079cbdfa3f6e9fc4114ad90"},"schema_version":"1.0","source":{"id":"2006.11274","kind":"arxiv","version":1}},"canonical_sha256":"20e0c8267fda45d19d8be2a71bb43552b872c682f54124718ac3d6f31ad8ce85","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"20e0c8267fda45d19d8be2a71bb43552b872c682f54124718ac3d6f31ad8ce85","first_computed_at":"2026-07-05T01:11:37.003538Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T01:11:37.003538Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"vTtfwZSH8Xc1kNOFISzFlS/AihphT4QvZED7aHxs/gQMq+bDL4P6wLgbWqYKMCeXB2EoBBuqLmT+Q8c5yDPcCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T01:11:37.003964Z","signed_message":"canonical_sha256_bytes"},"source_id":"2006.11274","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:1fe12b99a426f4e189bead445cbaf7243637de2e90ecb49d5317fb46f7a7aaa3","sha256:e67a963ca837b1f59f219327a190ba82a13c140938727f40b3a5cc837e4c86a8"],"state_sha256":"428171d6f45667b012ed91387e92ce51f0239475876803f479d1777990fcb6b4"}