{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:AT4OJY6TWKOSMUBH3G67JKV3XX","short_pith_number":"pith:AT4OJY6T","schema_version":"1.0","canonical_sha256":"04f8e4e3d3b29d265027d9bdf4aabbbdd6f5a0c8052256768edebb8fe226a003","source":{"kind":"arxiv","id":"2010.05901","version":2},"attestation_state":"computed","paper":{"title":"Nearly Minimax Optimal Reward-free Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Simon S. Du, Xiangyang Ji, Zihan Zhang","submitted_at":"2020-10-12T17:51:19Z","abstract_excerpt":"We study the reward-free reinforcement learning framework, which is particularly suitable for batch reinforcement learning and scenarios where one needs policies for multiple reward functions. This framework has two phases. In the exploration phase, the agent collects trajectories by interacting with the environment without using any reward signal. In the planning phase, the agent needs to return a near-optimal policy for arbitrary reward functions. We give a new efficient algorithm, \\textbf{S}taged \\textbf{S}ampling + \\textbf{T}runcated \\textbf{P}lanning (\\algoname), which interacts with the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.05901","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-10-12T17:51:19Z","cross_cats_sorted":[],"title_canon_sha256":"0ac4287a1a77977b1fabdf403dc9af859e0376fe5c172552a352505c94a88156","abstract_canon_sha256":"3c17258ccdd58798cf76a7dfbeaf02e453bfc276fa2326508b6966530c2a7bff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:45:31.038714Z","signature_b64":"tz86rjCNgSydY69j/MmIxgIaAqRSSC+of8U23Tu3Yeor5fUtf57i13njjwz5L9cpAWHwOUWertq7uawPZgRSDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"04f8e4e3d3b29d265027d9bdf4aabbbdd6f5a0c8052256768edebb8fe226a003","last_reissued_at":"2026-07-05T01:45:31.038331Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:45:31.038331Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Nearly Minimax Optimal Reward-free Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Simon S. Du, Xiangyang Ji, Zihan Zhang","submitted_at":"2020-10-12T17:51:19Z","abstract_excerpt":"We study the reward-free reinforcement learning framework, which is particularly suitable for batch reinforcement learning and scenarios where one needs policies for multiple reward functions. This framework has two phases. In the exploration phase, the agent collects trajectories by interacting with the environment without using any reward signal. In the planning phase, the agent needs to return a near-optimal policy for arbitrary reward functions. We give a new efficient algorithm, \\textbf{S}taged \\textbf{S}ampling + \\textbf{T}runcated \\textbf{P}lanning (\\algoname), which interacts with the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.05901","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.05901/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.05901","created_at":"2026-07-05T01:45:31.038387+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.05901v2","created_at":"2026-07-05T01:45:31.038387+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.05901","created_at":"2026-07-05T01:45:31.038387+00:00"},{"alias_kind":"pith_short_12","alias_value":"AT4OJY6TWKOS","created_at":"2026-07-05T01:45:31.038387+00:00"},{"alias_kind":"pith_short_16","alias_value":"AT4OJY6TWKOSMUBH","created_at":"2026-07-05T01:45:31.038387+00:00"},{"alias_kind":"pith_short_8","alias_value":"AT4OJY6T","created_at":"2026-07-05T01:45:31.038387+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.03891","citing_title":"Provable Multi-Task Reinforcement Learning: A Representation Learning Framework with Low Rank Rewards","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01242","citing_title":"Breaking the Computational Barrier: Provably Efficient Actor-Critic for Low-Rank MDPs","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AT4OJY6TWKOSMUBH3G67JKV3XX","json":"https://pith.science/pith/AT4OJY6TWKOSMUBH3G67JKV3XX.json","graph_json":"https://pith.science/api/pith-number/AT4OJY6TWKOSMUBH3G67JKV3XX/graph.json","events_json":"https://pith.science/api/pith-number/AT4OJY6TWKOSMUBH3G67JKV3XX/events.json","paper":"https://pith.science/paper/AT4OJY6T"},"agent_actions":{"view_html":"https://pith.science/pith/AT4OJY6TWKOSMUBH3G67JKV3XX","download_json":"https://pith.science/pith/AT4OJY6TWKOSMUBH3G67JKV3XX.json","view_paper":"https://pith.science/paper/AT4OJY6T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.05901&json=true","fetch_graph":"https://pith.science/api/pith-number/AT4OJY6TWKOSMUBH3G67JKV3XX/graph.json","fetch_events":"https://pith.science/api/pith-number/AT4OJY6TWKOSMUBH3G67JKV3XX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AT4OJY6TWKOSMUBH3G67JKV3XX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AT4OJY6TWKOSMUBH3G67JKV3XX/action/storage_attestation","attest_author":"https://pith.science/pith/AT4OJY6TWKOSMUBH3G67JKV3XX/action/author_attestation","sign_citation":"https://pith.science/pith/AT4OJY6TWKOSMUBH3G67JKV3XX/action/citation_signature","submit_replication":"https://pith.science/pith/AT4OJY6TWKOSMUBH3G67JKV3XX/action/replication_record"}},"created_at":"2026-07-05T01:45:31.038387+00:00","updated_at":"2026-07-05T01:45:31.038387+00:00"}