{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BBHYB2ZLFKBPGRD3X67Z2CVF65","short_pith_number":"pith:BBHYB2ZL","schema_version":"1.0","canonical_sha256":"084f80eb2b2a82f3447bbfbf9d0aa5f75b0054e4fdf041d87ba80f59661e278e","source":{"kind":"arxiv","id":"2303.02215","version":1},"attestation_state":"computed","paper":{"title":"Learning Stabilization Control from Observations by Learning Lyapunov-like Proxy Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Chiaki Hirayama, Milan Ganai, Sicun Gao, Ya-Chien Chang","submitted_at":"2023-03-03T21:13:27Z","abstract_excerpt":"The deployment of Reinforcement Learning to robotics applications faces the difficulty of reward engineering. Therefore, approaches have focused on creating reward functions by Learning from Observations (LfO) which is the task of learning policies from expert trajectories that only contain state sequences. We propose new methods for LfO for the important class of continuous control problems of learning to stabilize, by introducing intermediate proxy models acting as reward functions between the expert and the agent policy based on Lyapunov stability theory. Our LfO training process consists o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.02215","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2023-03-03T21:13:27Z","cross_cats_sorted":[],"title_canon_sha256":"53b49e69278dd8f96a1d7d575a2edbe54bca4db9c2e112676b2d51aee834d404","abstract_canon_sha256":"f0dcc2b5d1b4ec290aef9c4ba37acdc9cc1cf5a4aacf931278b1e6671657c4dc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:48:08.105553Z","signature_b64":"WMgg2g4LMN6OJVuXaWoTyEfXF8R64R2YmJ830rndbL7cFglMdkcVl8KP5w/E7ZHRVl7r/2/u5TZmM1ubXPXTDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"084f80eb2b2a82f3447bbfbf9d0aa5f75b0054e4fdf041d87ba80f59661e278e","last_reissued_at":"2026-07-05T05:48:08.105076Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:48:08.105076Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Stabilization Control from Observations by Learning Lyapunov-like Proxy Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Chiaki Hirayama, Milan Ganai, Sicun Gao, Ya-Chien Chang","submitted_at":"2023-03-03T21:13:27Z","abstract_excerpt":"The deployment of Reinforcement Learning to robotics applications faces the difficulty of reward engineering. Therefore, approaches have focused on creating reward functions by Learning from Observations (LfO) which is the task of learning policies from expert trajectories that only contain state sequences. We propose new methods for LfO for the important class of continuous control problems of learning to stabilize, by introducing intermediate proxy models acting as reward functions between the expert and the agent policy based on Lyapunov stability theory. Our LfO training process consists o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.02215","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.02215/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.02215","created_at":"2026-07-05T05:48:08.105133+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.02215v1","created_at":"2026-07-05T05:48:08.105133+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.02215","created_at":"2026-07-05T05:48:08.105133+00:00"},{"alias_kind":"pith_short_12","alias_value":"BBHYB2ZLFKBP","created_at":"2026-07-05T05:48:08.105133+00:00"},{"alias_kind":"pith_short_16","alias_value":"BBHYB2ZLFKBPGRD3","created_at":"2026-07-05T05:48:08.105133+00:00"},{"alias_kind":"pith_short_8","alias_value":"BBHYB2ZL","created_at":"2026-07-05T05:48:08.105133+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BBHYB2ZLFKBPGRD3X67Z2CVF65","json":"https://pith.science/pith/BBHYB2ZLFKBPGRD3X67Z2CVF65.json","graph_json":"https://pith.science/api/pith-number/BBHYB2ZLFKBPGRD3X67Z2CVF65/graph.json","events_json":"https://pith.science/api/pith-number/BBHYB2ZLFKBPGRD3X67Z2CVF65/events.json","paper":"https://pith.science/paper/BBHYB2ZL"},"agent_actions":{"view_html":"https://pith.science/pith/BBHYB2ZLFKBPGRD3X67Z2CVF65","download_json":"https://pith.science/pith/BBHYB2ZLFKBPGRD3X67Z2CVF65.json","view_paper":"https://pith.science/paper/BBHYB2ZL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.02215&json=true","fetch_graph":"https://pith.science/api/pith-number/BBHYB2ZLFKBPGRD3X67Z2CVF65/graph.json","fetch_events":"https://pith.science/api/pith-number/BBHYB2ZLFKBPGRD3X67Z2CVF65/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BBHYB2ZLFKBPGRD3X67Z2CVF65/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BBHYB2ZLFKBPGRD3X67Z2CVF65/action/storage_attestation","attest_author":"https://pith.science/pith/BBHYB2ZLFKBPGRD3X67Z2CVF65/action/author_attestation","sign_citation":"https://pith.science/pith/BBHYB2ZLFKBPGRD3X67Z2CVF65/action/citation_signature","submit_replication":"https://pith.science/pith/BBHYB2ZLFKBPGRD3X67Z2CVF65/action/replication_record"}},"created_at":"2026-07-05T05:48:08.105133+00:00","updated_at":"2026-07-05T05:48:08.105133+00:00"}