{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4PNL5WFNJR3JHXC5R2OFWOBEXW","short_pith_number":"pith:4PNL5WFN","schema_version":"1.0","canonical_sha256":"e3dabed8ad4c7693dc5d8e9c5b3824bd86e00dd4242b1b8205225d8d9a1155dd","source":{"kind":"arxiv","id":"2406.00592","version":3},"attestation_state":"computed","paper":{"title":"Model Predictive Control and Reinforcement Learning: A Unified Framework Based on Dynamic Programming","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SY","math.OC"],"primary_cat":"eess.SY","authors_text":"Dimitri P. Bertsekas","submitted_at":"2024-06-02T02:01:03Z","abstract_excerpt":"In this paper we describe a new conceptual framework that connects approximate Dynamic Programming (DP), Model Predictive Control (MPC), and Reinforcement Learning (RL). This framework centers around two algorithms, which are designed largely independently of each other and operate in synergy through the powerful mechanism of Newton's method. We call them the off-line training and the on-line play algorithms. The names are borrowed from some of the major successes of RL involving games; primary examples are the recent (2017) AlphaZero program (which plays chess, [SHS17], [SSS17]), and the simi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.00592","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.SY","submitted_at":"2024-06-02T02:01:03Z","cross_cats_sorted":["cs.AI","cs.SY","math.OC"],"title_canon_sha256":"46731b2392ceadc2ba6087996af6eae09ef0e1d9ea9f500b27cd1e1c9ea3696b","abstract_canon_sha256":"896bb1c13f9a3dd7fb88363e0feebd6a77bd38912b6767d6a31533166db3c910"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:38:10.171169Z","signature_b64":"3uEi040JUecmUT/3Ay27em6d2D2LnM/Zoj6Y2OJIng7Vcx/ihnuH85/VCq4UUJsZQyTS6w1Du4Lpt/AFkDITCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e3dabed8ad4c7693dc5d8e9c5b3824bd86e00dd4242b1b8205225d8d9a1155dd","last_reissued_at":"2026-07-05T08:38:10.170683Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:38:10.170683Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model Predictive Control and Reinforcement Learning: A Unified Framework Based on Dynamic Programming","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SY","math.OC"],"primary_cat":"eess.SY","authors_text":"Dimitri P. Bertsekas","submitted_at":"2024-06-02T02:01:03Z","abstract_excerpt":"In this paper we describe a new conceptual framework that connects approximate Dynamic Programming (DP), Model Predictive Control (MPC), and Reinforcement Learning (RL). This framework centers around two algorithms, which are designed largely independently of each other and operate in synergy through the powerful mechanism of Newton's method. We call them the off-line training and the on-line play algorithms. The names are borrowed from some of the major successes of RL involving games; primary examples are the recent (2017) AlphaZero program (which plays chess, [SHS17], [SSS17]), and the simi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.00592","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.00592/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.00592","created_at":"2026-07-05T08:38:10.170744+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.00592v3","created_at":"2026-07-05T08:38:10.170744+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.00592","created_at":"2026-07-05T08:38:10.170744+00:00"},{"alias_kind":"pith_short_12","alias_value":"4PNL5WFNJR3J","created_at":"2026-07-05T08:38:10.170744+00:00"},{"alias_kind":"pith_short_16","alias_value":"4PNL5WFNJR3JHXC5","created_at":"2026-07-05T08:38:10.170744+00:00"},{"alias_kind":"pith_short_8","alias_value":"4PNL5WFN","created_at":"2026-07-05T08:38:10.170744+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.20853","citing_title":"Geometry of Neural Reinforcement Learning in Continuous State and Action Spaces","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4PNL5WFNJR3JHXC5R2OFWOBEXW","json":"https://pith.science/pith/4PNL5WFNJR3JHXC5R2OFWOBEXW.json","graph_json":"https://pith.science/api/pith-number/4PNL5WFNJR3JHXC5R2OFWOBEXW/graph.json","events_json":"https://pith.science/api/pith-number/4PNL5WFNJR3JHXC5R2OFWOBEXW/events.json","paper":"https://pith.science/paper/4PNL5WFN"},"agent_actions":{"view_html":"https://pith.science/pith/4PNL5WFNJR3JHXC5R2OFWOBEXW","download_json":"https://pith.science/pith/4PNL5WFNJR3JHXC5R2OFWOBEXW.json","view_paper":"https://pith.science/paper/4PNL5WFN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.00592&json=true","fetch_graph":"https://pith.science/api/pith-number/4PNL5WFNJR3JHXC5R2OFWOBEXW/graph.json","fetch_events":"https://pith.science/api/pith-number/4PNL5WFNJR3JHXC5R2OFWOBEXW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4PNL5WFNJR3JHXC5R2OFWOBEXW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4PNL5WFNJR3JHXC5R2OFWOBEXW/action/storage_attestation","attest_author":"https://pith.science/pith/4PNL5WFNJR3JHXC5R2OFWOBEXW/action/author_attestation","sign_citation":"https://pith.science/pith/4PNL5WFNJR3JHXC5R2OFWOBEXW/action/citation_signature","submit_replication":"https://pith.science/pith/4PNL5WFNJR3JHXC5R2OFWOBEXW/action/replication_record"}},"created_at":"2026-07-05T08:38:10.170744+00:00","updated_at":"2026-07-05T08:38:10.170744+00:00"}