{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:FYIKN2JAIIAZBTDYOMJMPQUODM","short_pith_number":"pith:FYIKN2JA","schema_version":"1.0","canonical_sha256":"2e10a6e920420190cc787312c7c28e1b31b6a83f633277b18b0ef4cfc33f4d71","source":{"kind":"arxiv","id":"2011.04021","version":2},"attestation_state":"computed","paper":{"title":"On the role of planning in model-based deep reinforcement learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Abram L. Friesen, Arthur Guez, Fabio Viola, Feryal Behbahani, Jessica B. Hamrick, Lars Buesing, Petar Veli\\v{c}kovi\\'c, Sims Witherspoon, Th\\'eophane Weber, Thomas Anthony","submitted_at":"2020-11-08T16:55:16Z","abstract_excerpt":"Model-based planning is often thought to be necessary for deep, careful reasoning and generalization in artificial agents. While recent successes of model-based reinforcement learning (MBRL) with deep function approximation have strengthened this hypothesis, the resulting diversity of model-based methods has also made it difficult to track which components drive success and why. In this paper, we seek to disentangle the contributions of recent methods by focusing on three questions: (1) How does planning benefit MBRL agents? (2) Within planning, what choices drive performance? (3) To what exte"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2011.04021","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-11-08T16:55:16Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"c37e7f1fa8bcdaaf3ce64e5dd2554cb644bc70429b05e5c94a735e01fd44754c","abstract_canon_sha256":"0c6f79abb5b3c53e7d719cecbc0cb39973afc2867062f615b409719679852d0a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:23:59.522006Z","signature_b64":"LKRHMttbRc4aLMQQbKp8FB2V5+Fo+J5umM+ftx+5K5nDWQm6rH5f/cpc4GTzqhKki17I2srbnRwtdm/OFbe2CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2e10a6e920420190cc787312c7c28e1b31b6a83f633277b18b0ef4cfc33f4d71","last_reissued_at":"2026-07-05T02:23:59.521502Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:23:59.521502Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the role of planning in model-based deep reinforcement learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Abram L. Friesen, Arthur Guez, Fabio Viola, Feryal Behbahani, Jessica B. Hamrick, Lars Buesing, Petar Veli\\v{c}kovi\\'c, Sims Witherspoon, Th\\'eophane Weber, Thomas Anthony","submitted_at":"2020-11-08T16:55:16Z","abstract_excerpt":"Model-based planning is often thought to be necessary for deep, careful reasoning and generalization in artificial agents. While recent successes of model-based reinforcement learning (MBRL) with deep function approximation have strengthened this hypothesis, the resulting diversity of model-based methods has also made it difficult to track which components drive success and why. In this paper, we seek to disentangle the contributions of recent methods by focusing on three questions: (1) How does planning benefit MBRL agents? (2) Within planning, what choices drive performance? (3) To what exte"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2011.04021","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2011.04021/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2011.04021","created_at":"2026-07-05T02:23:59.521561+00:00"},{"alias_kind":"arxiv_version","alias_value":"2011.04021v2","created_at":"2026-07-05T02:23:59.521561+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2011.04021","created_at":"2026-07-05T02:23:59.521561+00:00"},{"alias_kind":"pith_short_12","alias_value":"FYIKN2JAIIAZ","created_at":"2026-07-05T02:23:59.521561+00:00"},{"alias_kind":"pith_short_16","alias_value":"FYIKN2JAIIAZBTDY","created_at":"2026-07-05T02:23:59.521561+00:00"},{"alias_kind":"pith_short_8","alias_value":"FYIKN2JA","created_at":"2026-07-05T02:23:59.521561+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26463","citing_title":"Finding the Time to Think: Learning Planning Budgets in Real-Time RL","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26463","citing_title":"Finding the Time to Think: Learning Planning Budgets in Real-Time RL","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03413","citing_title":"Learning to Theorize the World from Observation","ref_index":122,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02777","citing_title":"Decoupled Guidance Diffusion for Adaptive Offline Safe Reinforcement Learning","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FYIKN2JAIIAZBTDYOMJMPQUODM","json":"https://pith.science/pith/FYIKN2JAIIAZBTDYOMJMPQUODM.json","graph_json":"https://pith.science/api/pith-number/FYIKN2JAIIAZBTDYOMJMPQUODM/graph.json","events_json":"https://pith.science/api/pith-number/FYIKN2JAIIAZBTDYOMJMPQUODM/events.json","paper":"https://pith.science/paper/FYIKN2JA"},"agent_actions":{"view_html":"https://pith.science/pith/FYIKN2JAIIAZBTDYOMJMPQUODM","download_json":"https://pith.science/pith/FYIKN2JAIIAZBTDYOMJMPQUODM.json","view_paper":"https://pith.science/paper/FYIKN2JA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2011.04021&json=true","fetch_graph":"https://pith.science/api/pith-number/FYIKN2JAIIAZBTDYOMJMPQUODM/graph.json","fetch_events":"https://pith.science/api/pith-number/FYIKN2JAIIAZBTDYOMJMPQUODM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FYIKN2JAIIAZBTDYOMJMPQUODM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FYIKN2JAIIAZBTDYOMJMPQUODM/action/storage_attestation","attest_author":"https://pith.science/pith/FYIKN2JAIIAZBTDYOMJMPQUODM/action/author_attestation","sign_citation":"https://pith.science/pith/FYIKN2JAIIAZBTDYOMJMPQUODM/action/citation_signature","submit_replication":"https://pith.science/pith/FYIKN2JAIIAZBTDYOMJMPQUODM/action/replication_record"}},"created_at":"2026-07-05T02:23:59.521561+00:00","updated_at":"2026-07-05T02:23:59.521561+00:00"}