{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:E27NAPFHVCJ6QKPDBMAYVIIQHG","short_pith_number":"pith:E27NAPFH","schema_version":"1.0","canonical_sha256":"26bed03ca7a893e829e30b018aa110399e74c1e688831c25e3e18b034aa50ef6","source":{"kind":"arxiv","id":"2201.09765","version":2},"attestation_state":"computed","paper":{"title":"Generative Planning for Temporally Coordinated Exploration in Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Haichao Zhang, Haonan Yu, Wei Xu","submitted_at":"2022-01-24T15:53:32Z","abstract_excerpt":"Standard model-free reinforcement learning algorithms optimize a policy that generates the action to be taken in the current time step in order to maximize expected future return. While flexible, it faces difficulties arising from the inefficient exploration due to its single step nature. In this work, we present Generative Planning method (GPM), which can generate actions not only for the current step, but also for a number of future steps (thus termed as generative planning). This brings several benefits to GPM. Firstly, since GPM is trained by maximizing value, the plans generated from it c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.09765","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-01-24T15:53:32Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"31f26a2f4414f24a86e446fd32d91f6fe26acf3186f416bc603ff9c984ef4dd7","abstract_canon_sha256":"86d92d18245b083f6cfa7773de3752ad14bdf7aef04724b7afabedc1a2b97532"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:54:04.528480Z","signature_b64":"vFfdTdwzkQ5nKdK12/J7IIea4Bxj1j4SBYKZUW8znf7BlOYuA/X7BaovU+Yk3PN1VUSCndj8AgPtREid+6zqBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"26bed03ca7a893e829e30b018aa110399e74c1e688831c25e3e18b034aa50ef6","last_reissued_at":"2026-07-05T03:54:04.528083Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:54:04.528083Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generative Planning for Temporally Coordinated Exploration in Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Haichao Zhang, Haonan Yu, Wei Xu","submitted_at":"2022-01-24T15:53:32Z","abstract_excerpt":"Standard model-free reinforcement learning algorithms optimize a policy that generates the action to be taken in the current time step in order to maximize expected future return. While flexible, it faces difficulties arising from the inefficient exploration due to its single step nature. In this work, we present Generative Planning method (GPM), which can generate actions not only for the current step, but also for a number of future steps (thus termed as generative planning). This brings several benefits to GPM. Firstly, since GPM is trained by maximizing value, the plans generated from it c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.09765","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.09765/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.09765","created_at":"2026-07-05T03:54:04.528137+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.09765v2","created_at":"2026-07-05T03:54:04.528137+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.09765","created_at":"2026-07-05T03:54:04.528137+00:00"},{"alias_kind":"pith_short_12","alias_value":"E27NAPFHVCJ6","created_at":"2026-07-05T03:54:04.528137+00:00"},{"alias_kind":"pith_short_16","alias_value":"E27NAPFHVCJ6QKPD","created_at":"2026-07-05T03:54:04.528137+00:00"},{"alias_kind":"pith_short_8","alias_value":"E27NAPFH","created_at":"2026-07-05T03:54:04.528137+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.15308","citing_title":"RAD-2: Scaling Reinforcement Learning in a Generator-Discriminator Framework","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E27NAPFHVCJ6QKPDBMAYVIIQHG","json":"https://pith.science/pith/E27NAPFHVCJ6QKPDBMAYVIIQHG.json","graph_json":"https://pith.science/api/pith-number/E27NAPFHVCJ6QKPDBMAYVIIQHG/graph.json","events_json":"https://pith.science/api/pith-number/E27NAPFHVCJ6QKPDBMAYVIIQHG/events.json","paper":"https://pith.science/paper/E27NAPFH"},"agent_actions":{"view_html":"https://pith.science/pith/E27NAPFHVCJ6QKPDBMAYVIIQHG","download_json":"https://pith.science/pith/E27NAPFHVCJ6QKPDBMAYVIIQHG.json","view_paper":"https://pith.science/paper/E27NAPFH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.09765&json=true","fetch_graph":"https://pith.science/api/pith-number/E27NAPFHVCJ6QKPDBMAYVIIQHG/graph.json","fetch_events":"https://pith.science/api/pith-number/E27NAPFHVCJ6QKPDBMAYVIIQHG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E27NAPFHVCJ6QKPDBMAYVIIQHG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E27NAPFHVCJ6QKPDBMAYVIIQHG/action/storage_attestation","attest_author":"https://pith.science/pith/E27NAPFHVCJ6QKPDBMAYVIIQHG/action/author_attestation","sign_citation":"https://pith.science/pith/E27NAPFHVCJ6QKPDBMAYVIIQHG/action/citation_signature","submit_replication":"https://pith.science/pith/E27NAPFHVCJ6QKPDBMAYVIIQHG/action/replication_record"}},"created_at":"2026-07-05T03:54:04.528137+00:00","updated_at":"2026-07-05T03:54:04.528137+00:00"}