{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:NYJHYOBHRMMFFKMF4TBTXX7HSA","short_pith_number":"pith:NYJHYOBH","schema_version":"1.0","canonical_sha256":"6e127c38278b1852a985e4c33bdfe79000b42e9cd7329866c94cb9c80d21219c","source":{"kind":"arxiv","id":"1906.08649","version":1},"attestation_state":"computed","paper":{"title":"Exploring Model-based Planning with Policy Networks","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jimmy Ba, Tingwu Wang","submitted_at":"2019-06-20T14:13:12Z","abstract_excerpt":"Model-based reinforcement learning (MBRL) with model-predictive control or online planning has shown great potential for locomotion control tasks in terms of both sample efficiency and asymptotic performance. Despite their initial successes, the existing planning methods search from candidate sequences randomly generated in the action space, which is inefficient in complex high-dimensional environments. In this paper, we propose a novel MBRL algorithm, model-based policy planning (POPLIN), that combines policy networks with online planning. More specifically, we formulate action planning at ea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1906.08649","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2019-06-20T14:13:12Z","cross_cats_sorted":["cs.AI","cs.RO","stat.ML"],"title_canon_sha256":"18b1fcedcb4cd2060009a4fb7d56fc6aa95e6320e1dffca00c457e68419b2ac6","abstract_canon_sha256":"116d9583a5a50d5bcd891c570bf58631d190b2058f294346da5c01e0a9f20cb6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-17T23:42:49.977824Z","signature_b64":"Ls0Hfr3MkzFOBpiLSfps7lQHh+eohBRKe9ki3q6YBsx+A0rXaWEVSiYqKnzsEzqLDpoYtfqrgmzKbu/LbgwQCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6e127c38278b1852a985e4c33bdfe79000b42e9cd7329866c94cb9c80d21219c","last_reissued_at":"2026-05-17T23:42:49.977442Z","signature_status":"signed_v1","first_computed_at":"2026-05-17T23:42:49.977442Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring Model-based Planning with Policy Networks","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jimmy Ba, Tingwu Wang","submitted_at":"2019-06-20T14:13:12Z","abstract_excerpt":"Model-based reinforcement learning (MBRL) with model-predictive control or online planning has shown great potential for locomotion control tasks in terms of both sample efficiency and asymptotic performance. Despite their initial successes, the existing planning methods search from candidate sequences randomly generated in the action space, which is inefficient in complex high-dimensional environments. In this paper, we propose a novel MBRL algorithm, model-based policy planning (POPLIN), that combines policy networks with online planning. More specifically, we formulate action planning at ea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1906.08649","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1906.08649","created_at":"2026-05-17T23:42:49.977499+00:00"},{"alias_kind":"arxiv_version","alias_value":"1906.08649v1","created_at":"2026-05-17T23:42:49.977499+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1906.08649","created_at":"2026-05-17T23:42:49.977499+00:00"},{"alias_kind":"pith_short_12","alias_value":"NYJHYOBHRMMF","created_at":"2026-05-18T12:33:24.271573+00:00"},{"alias_kind":"pith_short_16","alias_value":"NYJHYOBHRMMFFKMF","created_at":"2026-05-18T12:33:24.271573+00:00"},{"alias_kind":"pith_short_8","alias_value":"NYJHYOBH","created_at":"2026-05-18T12:33:24.271573+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2606.10825","citing_title":"MODIP: Efficient Model-Based Optimization for Diffusion Policies","ref_index":40,"is_internal_anchor":true},{"citing_arxiv_id":"2010.02193","citing_title":"Mastering Atari with Discrete World Models","ref_index":48,"is_internal_anchor":true},{"citing_arxiv_id":"1912.01603","citing_title":"Dream to Control: Learning Behaviors by Latent Imagination","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06497","citing_title":"Hyperfastrl: Hypernetwork-based reinforcement learning for unified control of parametric chaotic PDEs","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08958","citing_title":"WOMBET: World Model-Based Experience Transfer for Robust and Sample-efficient Reinforcement Learning","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NYJHYOBHRMMFFKMF4TBTXX7HSA","json":"https://pith.science/pith/NYJHYOBHRMMFFKMF4TBTXX7HSA.json","graph_json":"https://pith.science/api/pith-number/NYJHYOBHRMMFFKMF4TBTXX7HSA/graph.json","events_json":"https://pith.science/api/pith-number/NYJHYOBHRMMFFKMF4TBTXX7HSA/events.json","paper":"https://pith.science/paper/NYJHYOBH"},"agent_actions":{"view_html":"https://pith.science/pith/NYJHYOBHRMMFFKMF4TBTXX7HSA","download_json":"https://pith.science/pith/NYJHYOBHRMMFFKMF4TBTXX7HSA.json","view_paper":"https://pith.science/paper/NYJHYOBH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1906.08649&json=true","fetch_graph":"https://pith.science/api/pith-number/NYJHYOBHRMMFFKMF4TBTXX7HSA/graph.json","fetch_events":"https://pith.science/api/pith-number/NYJHYOBHRMMFFKMF4TBTXX7HSA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NYJHYOBHRMMFFKMF4TBTXX7HSA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NYJHYOBHRMMFFKMF4TBTXX7HSA/action/storage_attestation","attest_author":"https://pith.science/pith/NYJHYOBHRMMFFKMF4TBTXX7HSA/action/author_attestation","sign_citation":"https://pith.science/pith/NYJHYOBHRMMFFKMF4TBTXX7HSA/action/citation_signature","submit_replication":"https://pith.science/pith/NYJHYOBHRMMFFKMF4TBTXX7HSA/action/replication_record"}},"created_at":"2026-05-17T23:42:49.977499+00:00","updated_at":"2026-05-17T23:42:49.977499+00:00"}