{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:MKETMLMZD5XWUEJXUILZDN6VQ2","short_pith_number":"pith:MKETMLMZ","schema_version":"1.0","canonical_sha256":"6289362d991f6f6a1137a21791b7d586bd9e8ef75f920978bc060e5208fbf1ea","source":{"kind":"arxiv","id":"2206.14057","version":3},"attestation_state":"computed","paper":{"title":"Safe Exploration Incurs Nearly No Additional Sample Complexity for Reward-free RL","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Jing Yang, Ruiquan Huang, Yingbin Liang","submitted_at":"2022-06-28T15:00:45Z","abstract_excerpt":"Reward-free reinforcement learning (RF-RL), a recently introduced RL paradigm, relies on random action-taking to explore the unknown environment without any reward feedback information. While the primary goal of the exploration phase in RF-RL is to reduce the uncertainty in the estimated model with minimum number of trajectories, in practice, the agent often needs to abide by certain safety constraint at the same time. It remains unclear how such safe exploration requirement would affect the corresponding sample complexity in order to achieve the desired optimality of the obtained policy in pl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.14057","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-06-28T15:00:45Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"92ca9a2536e7b284c6f5e874b71a9310c10d5a78675d426a0d64832ef31d2844","abstract_canon_sha256":"f162fa22708e10bc0865f61345ff79eb926cf76a0ff99d3a565fcee6a76bbc7d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:53:32.470088Z","signature_b64":"TK54NeZJacDjEOaRBNvs8mJeUUsyi7lqmgWwTN/ZrJZiRChNztQGWp0brZ/eXITs5gZA7+m6r+7B/TPIMwY4CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6289362d991f6f6a1137a21791b7d586bd9e8ef75f920978bc060e5208fbf1ea","last_reissued_at":"2026-07-05T05:53:32.469597Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:53:32.469597Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safe Exploration Incurs Nearly No Additional Sample Complexity for Reward-free RL","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Jing Yang, Ruiquan Huang, Yingbin Liang","submitted_at":"2022-06-28T15:00:45Z","abstract_excerpt":"Reward-free reinforcement learning (RF-RL), a recently introduced RL paradigm, relies on random action-taking to explore the unknown environment without any reward feedback information. While the primary goal of the exploration phase in RF-RL is to reduce the uncertainty in the estimated model with minimum number of trajectories, in practice, the agent often needs to abide by certain safety constraint at the same time. It remains unclear how such safe exploration requirement would affect the corresponding sample complexity in order to achieve the desired optimality of the obtained policy in pl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.14057","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.14057/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.14057","created_at":"2026-07-05T05:53:32.469652+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.14057v3","created_at":"2026-07-05T05:53:32.469652+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.14057","created_at":"2026-07-05T05:53:32.469652+00:00"},{"alias_kind":"pith_short_12","alias_value":"MKETMLMZD5XW","created_at":"2026-07-05T05:53:32.469652+00:00"},{"alias_kind":"pith_short_16","alias_value":"MKETMLMZD5XWUEJX","created_at":"2026-07-05T05:53:32.469652+00:00"},{"alias_kind":"pith_short_8","alias_value":"MKETMLMZ","created_at":"2026-07-05T05:53:32.469652+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.01242","citing_title":"Breaking the Computational Barrier: Provably Efficient Actor-Critic for Low-Rank MDPs","ref_index":69,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MKETMLMZD5XWUEJXUILZDN6VQ2","json":"https://pith.science/pith/MKETMLMZD5XWUEJXUILZDN6VQ2.json","graph_json":"https://pith.science/api/pith-number/MKETMLMZD5XWUEJXUILZDN6VQ2/graph.json","events_json":"https://pith.science/api/pith-number/MKETMLMZD5XWUEJXUILZDN6VQ2/events.json","paper":"https://pith.science/paper/MKETMLMZ"},"agent_actions":{"view_html":"https://pith.science/pith/MKETMLMZD5XWUEJXUILZDN6VQ2","download_json":"https://pith.science/pith/MKETMLMZD5XWUEJXUILZDN6VQ2.json","view_paper":"https://pith.science/paper/MKETMLMZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.14057&json=true","fetch_graph":"https://pith.science/api/pith-number/MKETMLMZD5XWUEJXUILZDN6VQ2/graph.json","fetch_events":"https://pith.science/api/pith-number/MKETMLMZD5XWUEJXUILZDN6VQ2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MKETMLMZD5XWUEJXUILZDN6VQ2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MKETMLMZD5XWUEJXUILZDN6VQ2/action/storage_attestation","attest_author":"https://pith.science/pith/MKETMLMZD5XWUEJXUILZDN6VQ2/action/author_attestation","sign_citation":"https://pith.science/pith/MKETMLMZD5XWUEJXUILZDN6VQ2/action/citation_signature","submit_replication":"https://pith.science/pith/MKETMLMZD5XWUEJXUILZDN6VQ2/action/replication_record"}},"created_at":"2026-07-05T05:53:32.469652+00:00","updated_at":"2026-07-05T05:53:32.469652+00:00"}