{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TQMRDSKZ5X2X3NS3QELMGC2W6C","short_pith_number":"pith:TQMRDSKZ","schema_version":"1.0","canonical_sha256":"9c1911c959edf57db65b8116c30b56f088e2dca4e2e0d129e5f57babe2a15a1d","source":{"kind":"arxiv","id":"2306.09210","version":1},"attestation_state":"computed","paper":{"title":"Optimal Exploration for Model-Based RL in Nonlinear Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO","cs.SY","eess.SY","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Andrew Wagenmaker, Guanya Shi, Kevin Jamieson","submitted_at":"2023-06-15T15:47:50Z","abstract_excerpt":"Learning to control unknown nonlinear dynamical systems is a fundamental problem in reinforcement learning and control theory. A commonly applied approach is to first explore the environment (exploration), learn an accurate model of it (system identification), and then compute an optimal controller with the minimum cost on this estimated system (policy optimization). While existing work has shown that it is possible to learn a uniformly good model of the system~\\citep{mania2020active}, in practice, if we aim to learn a good controller with a low cost on the actual system, certain system parame"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.09210","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-15T15:47:50Z","cross_cats_sorted":["cs.RO","cs.SY","eess.SY","math.OC","stat.ML"],"title_canon_sha256":"546e5b9acfc4e3dddb09034f790636453e62b6770e1a3667f0c35ca393749469","abstract_canon_sha256":"92b62b371f6a928bffd5cd1bbd04dac302bf4bc2ae83da7ff6faec1a9a3e2239"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:21:09.765011Z","signature_b64":"V0qaQFB22LGIhYuBB8mKWXvAepK8Zv+kFpYi8FFYEKVf3yeoGUdw2aonPH7MK7vJ1cbUgZCGt5MJ3uY5SfyUDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9c1911c959edf57db65b8116c30b56f088e2dca4e2e0d129e5f57babe2a15a1d","last_reissued_at":"2026-07-05T06:21:09.764599Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:21:09.764599Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Optimal Exploration for Model-Based RL in Nonlinear Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO","cs.SY","eess.SY","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Andrew Wagenmaker, Guanya Shi, Kevin Jamieson","submitted_at":"2023-06-15T15:47:50Z","abstract_excerpt":"Learning to control unknown nonlinear dynamical systems is a fundamental problem in reinforcement learning and control theory. A commonly applied approach is to first explore the environment (exploration), learn an accurate model of it (system identification), and then compute an optimal controller with the minimum cost on this estimated system (policy optimization). While existing work has shown that it is possible to learn a uniformly good model of the system~\\citep{mania2020active}, in practice, if we aim to learn a good controller with a low cost on the actual system, certain system parame"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.09210","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.09210/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.09210","created_at":"2026-07-05T06:21:09.764656+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.09210v1","created_at":"2026-07-05T06:21:09.764656+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.09210","created_at":"2026-07-05T06:21:09.764656+00:00"},{"alias_kind":"pith_short_12","alias_value":"TQMRDSKZ5X2X","created_at":"2026-07-05T06:21:09.764656+00:00"},{"alias_kind":"pith_short_16","alias_value":"TQMRDSKZ5X2X3NS3","created_at":"2026-07-05T06:21:09.764656+00:00"},{"alias_kind":"pith_short_8","alias_value":"TQMRDSKZ","created_at":"2026-07-05T06:21:09.764656+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TQMRDSKZ5X2X3NS3QELMGC2W6C","json":"https://pith.science/pith/TQMRDSKZ5X2X3NS3QELMGC2W6C.json","graph_json":"https://pith.science/api/pith-number/TQMRDSKZ5X2X3NS3QELMGC2W6C/graph.json","events_json":"https://pith.science/api/pith-number/TQMRDSKZ5X2X3NS3QELMGC2W6C/events.json","paper":"https://pith.science/paper/TQMRDSKZ"},"agent_actions":{"view_html":"https://pith.science/pith/TQMRDSKZ5X2X3NS3QELMGC2W6C","download_json":"https://pith.science/pith/TQMRDSKZ5X2X3NS3QELMGC2W6C.json","view_paper":"https://pith.science/paper/TQMRDSKZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.09210&json=true","fetch_graph":"https://pith.science/api/pith-number/TQMRDSKZ5X2X3NS3QELMGC2W6C/graph.json","fetch_events":"https://pith.science/api/pith-number/TQMRDSKZ5X2X3NS3QELMGC2W6C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TQMRDSKZ5X2X3NS3QELMGC2W6C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TQMRDSKZ5X2X3NS3QELMGC2W6C/action/storage_attestation","attest_author":"https://pith.science/pith/TQMRDSKZ5X2X3NS3QELMGC2W6C/action/author_attestation","sign_citation":"https://pith.science/pith/TQMRDSKZ5X2X3NS3QELMGC2W6C/action/citation_signature","submit_replication":"https://pith.science/pith/TQMRDSKZ5X2X3NS3QELMGC2W6C/action/replication_record"}},"created_at":"2026-07-05T06:21:09.764656+00:00","updated_at":"2026-07-05T06:21:09.764656+00:00"}