{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:4VJDD6XEPVR6QWN2VEYG7DVF5D","short_pith_number":"pith:4VJDD6XE","schema_version":"1.0","canonical_sha256":"e55231fae47d63e859baa9306f8ea5e8db34dc23009350c53b65a6054fcc7d35","source":{"kind":"arxiv","id":"2109.00157","version":2},"attestation_state":"computed","paper":{"title":"A Survey of Exploration Methods in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Doina Precup, Harsh Satija, Herke van Hoof, Maziar Gomrokchi, Susan Amin","submitted_at":"2021-09-01T02:36:14Z","abstract_excerpt":"Exploration is an essential component of reinforcement learning algorithms, where agents need to learn how to predict and control unknown and often stochastic environments. Reinforcement learning agents depend crucially on exploration to obtain informative data for the learning process as the lack of enough information could hinder effective learning. In this article, we provide a survey of modern exploration methods in (Sequential) reinforcement learning, as well as a taxonomy of exploration methods."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.00157","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-09-01T02:36:14Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"415b397fa48eaffb380ecf767adc1c4dabd985bcc2a67802b919c636cc3ede17","abstract_canon_sha256":"2dbf5ceb5b8f069f1ff4698edddc08861bf2cb2c9c5fdd8e19931ccc9f14bf03"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:10:55.645142Z","signature_b64":"jdTY4yTJJKQLhedac8I4Ej8fDLN2sGOwEuyofcktkhOViCpBrYLNiaoAUAN1WL0sBLdAduTJEhKM/S/tJkUjCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e55231fae47d63e859baa9306f8ea5e8db34dc23009350c53b65a6054fcc7d35","last_reissued_at":"2026-07-05T03:10:55.644711Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:10:55.644711Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey of Exploration Methods in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Doina Precup, Harsh Satija, Herke van Hoof, Maziar Gomrokchi, Susan Amin","submitted_at":"2021-09-01T02:36:14Z","abstract_excerpt":"Exploration is an essential component of reinforcement learning algorithms, where agents need to learn how to predict and control unknown and often stochastic environments. Reinforcement learning agents depend crucially on exploration to obtain informative data for the learning process as the lack of enough information could hinder effective learning. In this article, we provide a survey of modern exploration methods in (Sequential) reinforcement learning, as well as a taxonomy of exploration methods."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.00157","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.00157/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.00157","created_at":"2026-07-05T03:10:55.644769+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.00157v2","created_at":"2026-07-05T03:10:55.644769+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.00157","created_at":"2026-07-05T03:10:55.644769+00:00"},{"alias_kind":"pith_short_12","alias_value":"4VJDD6XEPVR6","created_at":"2026-07-05T03:10:55.644769+00:00"},{"alias_kind":"pith_short_16","alias_value":"4VJDD6XEPVR6QWN2","created_at":"2026-07-05T03:10:55.644769+00:00"},{"alias_kind":"pith_short_8","alias_value":"4VJDD6XE","created_at":"2026-07-05T03:10:55.644769+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2601.22235","citing_title":"Smart Walkers in Discrete Space","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04207","citing_title":"Optimal Semiparametric Dynamic Pricing with Feature Diversity","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05820","citing_title":"SpecRL: Reinforcement Learning with Test-Based Completeness Rewards for Formal Specification Synthesis","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12645","citing_title":"Contextual Multi-Task Reinforcement Learning for Autonomous Reef Monitoring","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15614","citing_title":"Flexible Empowerment at Reasoning with Extended Best-of-N Sampling","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4VJDD6XEPVR6QWN2VEYG7DVF5D","json":"https://pith.science/pith/4VJDD6XEPVR6QWN2VEYG7DVF5D.json","graph_json":"https://pith.science/api/pith-number/4VJDD6XEPVR6QWN2VEYG7DVF5D/graph.json","events_json":"https://pith.science/api/pith-number/4VJDD6XEPVR6QWN2VEYG7DVF5D/events.json","paper":"https://pith.science/paper/4VJDD6XE"},"agent_actions":{"view_html":"https://pith.science/pith/4VJDD6XEPVR6QWN2VEYG7DVF5D","download_json":"https://pith.science/pith/4VJDD6XEPVR6QWN2VEYG7DVF5D.json","view_paper":"https://pith.science/paper/4VJDD6XE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.00157&json=true","fetch_graph":"https://pith.science/api/pith-number/4VJDD6XEPVR6QWN2VEYG7DVF5D/graph.json","fetch_events":"https://pith.science/api/pith-number/4VJDD6XEPVR6QWN2VEYG7DVF5D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4VJDD6XEPVR6QWN2VEYG7DVF5D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4VJDD6XEPVR6QWN2VEYG7DVF5D/action/storage_attestation","attest_author":"https://pith.science/pith/4VJDD6XEPVR6QWN2VEYG7DVF5D/action/author_attestation","sign_citation":"https://pith.science/pith/4VJDD6XEPVR6QWN2VEYG7DVF5D/action/citation_signature","submit_replication":"https://pith.science/pith/4VJDD6XEPVR6QWN2VEYG7DVF5D/action/replication_record"}},"created_at":"2026-07-05T03:10:55.644769+00:00","updated_at":"2026-07-05T03:10:55.644769+00:00"}