{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MYESUJJZSH4YVDZCLJINXCMAMO","short_pith_number":"pith:MYESUJJZ","schema_version":"1.0","canonical_sha256":"66092a253991f98a8f225a50db89806389478d09ee217cd84e4771d4e57bd49d","source":{"kind":"arxiv","id":"2410.16821","version":2},"attestation_state":"computed","paper":{"title":"Guiding Reinforcement Learning with Incomplete System Dynamics","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.SY","eess.SY"],"primary_cat":"cs.RO","authors_text":"Jingliang Duan, Lixian Zhang, Michael G. Forbes, Nathan P. Lawrence, Philip D. Loewen, R. Bhushan Gopaluni, Shuyuan Wang","submitted_at":"2024-10-22T08:48:48Z","abstract_excerpt":"Model-free reinforcement learning (RL) is inherently a reactive method, operating under the assumption that it starts with no prior knowledge of the system and entirely depends on trial-and-error for learning. This approach faces several challenges, such as poor sample efficiency, generalization, and the need for well-designed reward functions to guide learning effectively. On the other hand, controllers based on complete system dynamics do not require data. This paper addresses the intermediate situation where there is not enough model information for complete controller design, but there is "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.16821","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.RO","submitted_at":"2024-10-22T08:48:48Z","cross_cats_sorted":["cs.SY","eess.SY"],"title_canon_sha256":"d250bb4465636cac4e9d36c2c8503300333c3e608a92c60d66c70e1025f15111","abstract_canon_sha256":"4013d91e501d0d63def77db406b478127e62300d6e6f5329f0c8c708eb9d8bd2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:25:02.831722Z","signature_b64":"XopT8gcOD697HNdGCCobCRDuemsNAdir7ou5kPdErzaCOpF4et/ZRwAZNnjw7y0CtaFhBak9LMZKXqeNlRItAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"66092a253991f98a8f225a50db89806389478d09ee217cd84e4771d4e57bd49d","last_reissued_at":"2026-07-05T09:25:02.831220Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:25:02.831220Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Guiding Reinforcement Learning with Incomplete System Dynamics","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.SY","eess.SY"],"primary_cat":"cs.RO","authors_text":"Jingliang Duan, Lixian Zhang, Michael G. Forbes, Nathan P. Lawrence, Philip D. Loewen, R. Bhushan Gopaluni, Shuyuan Wang","submitted_at":"2024-10-22T08:48:48Z","abstract_excerpt":"Model-free reinforcement learning (RL) is inherently a reactive method, operating under the assumption that it starts with no prior knowledge of the system and entirely depends on trial-and-error for learning. This approach faces several challenges, such as poor sample efficiency, generalization, and the need for well-designed reward functions to guide learning effectively. On the other hand, controllers based on complete system dynamics do not require data. This paper addresses the intermediate situation where there is not enough model information for complete controller design, but there is "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.16821","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.16821/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.16821","created_at":"2026-07-05T09:25:02.831276+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.16821v2","created_at":"2026-07-05T09:25:02.831276+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.16821","created_at":"2026-07-05T09:25:02.831276+00:00"},{"alias_kind":"pith_short_12","alias_value":"MYESUJJZSH4Y","created_at":"2026-07-05T09:25:02.831276+00:00"},{"alias_kind":"pith_short_16","alias_value":"MYESUJJZSH4YVDZC","created_at":"2026-07-05T09:25:02.831276+00:00"},{"alias_kind":"pith_short_8","alias_value":"MYESUJJZ","created_at":"2026-07-05T09:25:02.831276+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.17473","citing_title":"DiLQR: Differentiable Iterative Linear Quadratic Regulator via Implicit Differentiation","ref_index":2024,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MYESUJJZSH4YVDZCLJINXCMAMO","json":"https://pith.science/pith/MYESUJJZSH4YVDZCLJINXCMAMO.json","graph_json":"https://pith.science/api/pith-number/MYESUJJZSH4YVDZCLJINXCMAMO/graph.json","events_json":"https://pith.science/api/pith-number/MYESUJJZSH4YVDZCLJINXCMAMO/events.json","paper":"https://pith.science/paper/MYESUJJZ"},"agent_actions":{"view_html":"https://pith.science/pith/MYESUJJZSH4YVDZCLJINXCMAMO","download_json":"https://pith.science/pith/MYESUJJZSH4YVDZCLJINXCMAMO.json","view_paper":"https://pith.science/paper/MYESUJJZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.16821&json=true","fetch_graph":"https://pith.science/api/pith-number/MYESUJJZSH4YVDZCLJINXCMAMO/graph.json","fetch_events":"https://pith.science/api/pith-number/MYESUJJZSH4YVDZCLJINXCMAMO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MYESUJJZSH4YVDZCLJINXCMAMO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MYESUJJZSH4YVDZCLJINXCMAMO/action/storage_attestation","attest_author":"https://pith.science/pith/MYESUJJZSH4YVDZCLJINXCMAMO/action/author_attestation","sign_citation":"https://pith.science/pith/MYESUJJZSH4YVDZCLJINXCMAMO/action/citation_signature","submit_replication":"https://pith.science/pith/MYESUJJZSH4YVDZCLJINXCMAMO/action/replication_record"}},"created_at":"2026-07-05T09:25:02.831276+00:00","updated_at":"2026-07-05T09:25:02.831276+00:00"}