{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:DPZTPJC65VQVKLFKGNE2XFWAKD","short_pith_number":"pith:DPZTPJC6","schema_version":"1.0","canonical_sha256":"1bf337a45eed61552caa3349ab96c050e2842ef55aeaac3451d9ac24caa78042","source":{"kind":"arxiv","id":"2010.10740","version":1},"attestation_state":"computed","paper":{"title":"Safety Verification of Model Based Reinforcement Learning Controllers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Akshita Gupta, Inseok Hwang","submitted_at":"2020-10-21T03:35:28Z","abstract_excerpt":"Model-based reinforcement learning (RL) has emerged as a promising tool for developing controllers for real world systems (e.g., robotics, autonomous driving, etc.). However, real systems often have constraints imposed on their state space which must be satisfied to ensure the safety of the system and its environment. Developing a verification tool for RL algorithms is challenging because the non-linear structure of neural networks impedes analytical verification of such models or controllers. To this end, we present a novel safety verification framework for model-based RL controllers using re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.10740","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-10-21T03:35:28Z","cross_cats_sorted":["cs.RO","cs.SY","eess.SY"],"title_canon_sha256":"be9a083d546c0aa9b6787ad7adb5091b4024203067ac11bac7d6321ad27180a6","abstract_canon_sha256":"cf5d9624c943bd329d2b0694c0c3c965d51364fcc7d61a3d6f0f6cb4b086b7b3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:44:54.046709Z","signature_b64":"+ZuoZXzs0ZNT/x7NFY3eI1OlDP4o+iK3r0NZIsWQ1VYurxip6MZuXvIwRk765UJqvNzTJbQPX/4CnWOHexbMAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1bf337a45eed61552caa3349ab96c050e2842ef55aeaac3451d9ac24caa78042","last_reissued_at":"2026-07-05T01:44:54.046276Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:44:54.046276Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safety Verification of Model Based Reinforcement Learning Controllers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Akshita Gupta, Inseok Hwang","submitted_at":"2020-10-21T03:35:28Z","abstract_excerpt":"Model-based reinforcement learning (RL) has emerged as a promising tool for developing controllers for real world systems (e.g., robotics, autonomous driving, etc.). However, real systems often have constraints imposed on their state space which must be satisfied to ensure the safety of the system and its environment. Developing a verification tool for RL algorithms is challenging because the non-linear structure of neural networks impedes analytical verification of such models or controllers. To this end, we present a novel safety verification framework for model-based RL controllers using re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.10740","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.10740/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.10740","created_at":"2026-07-05T01:44:54.046343+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.10740v1","created_at":"2026-07-05T01:44:54.046343+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.10740","created_at":"2026-07-05T01:44:54.046343+00:00"},{"alias_kind":"pith_short_12","alias_value":"DPZTPJC65VQV","created_at":"2026-07-05T01:44:54.046343+00:00"},{"alias_kind":"pith_short_16","alias_value":"DPZTPJC65VQVKLFK","created_at":"2026-07-05T01:44:54.046343+00:00"},{"alias_kind":"pith_short_8","alias_value":"DPZTPJC6","created_at":"2026-07-05T01:44:54.046343+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04812","citing_title":"Scenario Generation for Risk-Aware Reinforcement Learning with Probably Approximately Safe Guarantees","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DPZTPJC65VQVKLFKGNE2XFWAKD","json":"https://pith.science/pith/DPZTPJC65VQVKLFKGNE2XFWAKD.json","graph_json":"https://pith.science/api/pith-number/DPZTPJC65VQVKLFKGNE2XFWAKD/graph.json","events_json":"https://pith.science/api/pith-number/DPZTPJC65VQVKLFKGNE2XFWAKD/events.json","paper":"https://pith.science/paper/DPZTPJC6"},"agent_actions":{"view_html":"https://pith.science/pith/DPZTPJC65VQVKLFKGNE2XFWAKD","download_json":"https://pith.science/pith/DPZTPJC65VQVKLFKGNE2XFWAKD.json","view_paper":"https://pith.science/paper/DPZTPJC6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.10740&json=true","fetch_graph":"https://pith.science/api/pith-number/DPZTPJC65VQVKLFKGNE2XFWAKD/graph.json","fetch_events":"https://pith.science/api/pith-number/DPZTPJC65VQVKLFKGNE2XFWAKD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DPZTPJC65VQVKLFKGNE2XFWAKD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DPZTPJC65VQVKLFKGNE2XFWAKD/action/storage_attestation","attest_author":"https://pith.science/pith/DPZTPJC65VQVKLFKGNE2XFWAKD/action/author_attestation","sign_citation":"https://pith.science/pith/DPZTPJC65VQVKLFKGNE2XFWAKD/action/citation_signature","submit_replication":"https://pith.science/pith/DPZTPJC65VQVKLFKGNE2XFWAKD/action/replication_record"}},"created_at":"2026-07-05T01:44:54.046343+00:00","updated_at":"2026-07-05T01:44:54.046343+00:00"}