{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:OJLHTSKEUTAMJ2GYW7ACEN7YIH","short_pith_number":"pith:OJLHTSKE","schema_version":"1.0","canonical_sha256":"725679c944a4c0c4e8d8b7c02237f841c865e6f182a86db02c1e7b1e00bf3cd7","source":{"kind":"arxiv","id":"1906.02457","version":1},"attestation_state":"computed","paper":{"title":"Clustered Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Shen-Yi Zhao, Wu-Jun Li, Xiao Ma","submitted_at":"2019-06-06T07:35:02Z","abstract_excerpt":"Exploration strategy design is one of the challenging problems in reinforcement learning~(RL), especially when the environment contains a large state space or sparse rewards. During exploration, the agent tries to discover novel areas or high reward~(quality) areas. In most existing methods, the novelty and quality in the neighboring area of the current state are not well utilized to guide the exploration of the agent. To tackle this problem, we propose a novel RL framework, called \\underline{c}lustered \\underline{r}einforcement \\underline{l}earning~(CRL), for efficient exploration in RL. CRL "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1906.02457","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-06-06T07:35:02Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"8bcaaa5d6e55f620caffc70a11e714154a49f8a427cf1e4ee60cc570a761fb00","abstract_canon_sha256":"104897ad23b7d2f169f32e872a873bb07d6a6c5d2e59a07f1ed3e6012b1ce46a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-17T23:44:01.115763Z","signature_b64":"E+ETPpN99+0nKHR2xaDBTcz4P6YB2dzTorVA4b9fzLWogpLqp9gLC4TVz03UgI3/QXFivyfrgJmIl1CxqaiwCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"725679c944a4c0c4e8d8b7c02237f841c865e6f182a86db02c1e7b1e00bf3cd7","last_reissued_at":"2026-05-17T23:44:01.115136Z","signature_status":"signed_v1","first_computed_at":"2026-05-17T23:44:01.115136Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Clustered Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Shen-Yi Zhao, Wu-Jun Li, Xiao Ma","submitted_at":"2019-06-06T07:35:02Z","abstract_excerpt":"Exploration strategy design is one of the challenging problems in reinforcement learning~(RL), especially when the environment contains a large state space or sparse rewards. During exploration, the agent tries to discover novel areas or high reward~(quality) areas. In most existing methods, the novelty and quality in the neighboring area of the current state are not well utilized to guide the exploration of the agent. To tackle this problem, we propose a novel RL framework, called \\underline{c}lustered \\underline{r}einforcement \\underline{l}earning~(CRL), for efficient exploration in RL. CRL "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1906.02457","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1906.02457","created_at":"2026-05-17T23:44:01.115236+00:00"},{"alias_kind":"arxiv_version","alias_value":"1906.02457v1","created_at":"2026-05-17T23:44:01.115236+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1906.02457","created_at":"2026-05-17T23:44:01.115236+00:00"},{"alias_kind":"pith_short_12","alias_value":"OJLHTSKEUTAM","created_at":"2026-05-18T12:33:24.271573+00:00"},{"alias_kind":"pith_short_16","alias_value":"OJLHTSKEUTAMJ2GY","created_at":"2026-05-18T12:33:24.271573+00:00"},{"alias_kind":"pith_short_8","alias_value":"OJLHTSKE","created_at":"2026-05-18T12:33:24.271573+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OJLHTSKEUTAMJ2GYW7ACEN7YIH","json":"https://pith.science/pith/OJLHTSKEUTAMJ2GYW7ACEN7YIH.json","graph_json":"https://pith.science/api/pith-number/OJLHTSKEUTAMJ2GYW7ACEN7YIH/graph.json","events_json":"https://pith.science/api/pith-number/OJLHTSKEUTAMJ2GYW7ACEN7YIH/events.json","paper":"https://pith.science/paper/OJLHTSKE"},"agent_actions":{"view_html":"https://pith.science/pith/OJLHTSKEUTAMJ2GYW7ACEN7YIH","download_json":"https://pith.science/pith/OJLHTSKEUTAMJ2GYW7ACEN7YIH.json","view_paper":"https://pith.science/paper/OJLHTSKE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1906.02457&json=true","fetch_graph":"https://pith.science/api/pith-number/OJLHTSKEUTAMJ2GYW7ACEN7YIH/graph.json","fetch_events":"https://pith.science/api/pith-number/OJLHTSKEUTAMJ2GYW7ACEN7YIH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OJLHTSKEUTAMJ2GYW7ACEN7YIH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OJLHTSKEUTAMJ2GYW7ACEN7YIH/action/storage_attestation","attest_author":"https://pith.science/pith/OJLHTSKEUTAMJ2GYW7ACEN7YIH/action/author_attestation","sign_citation":"https://pith.science/pith/OJLHTSKEUTAMJ2GYW7ACEN7YIH/action/citation_signature","submit_replication":"https://pith.science/pith/OJLHTSKEUTAMJ2GYW7ACEN7YIH/action/replication_record"}},"created_at":"2026-05-17T23:44:01.115236+00:00","updated_at":"2026-05-17T23:44:01.115236+00:00"}