{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:LB5HLZ4MCQ3Y32RWEKV5RM246I","short_pith_number":"pith:LB5HLZ4M","schema_version":"1.0","canonical_sha256":"587a75e78c14378dea3622abd8b35cf231566cae17b16104adb409344621fbe5","source":{"kind":"arxiv","id":"2010.07968","version":2},"attestation_state":"computed","paper":{"title":"Constrained Model-based Reinforcement Learning with Robust Cross-Entropy Method","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.RO"],"primary_cat":"cs.AI","authors_text":"Baiming Chen, Ding Zhao, Hongyi Zhou, Martial Hebert, Sicheng Zhong, Zuxin Liu","submitted_at":"2020-10-15T18:19:35Z","abstract_excerpt":"This paper studies the constrained/safe reinforcement learning (RL) problem with sparse indicator signals for constraint violations. We propose a model-based approach to enable RL agents to effectively explore the environment with unknown system dynamics and environment constraints given a significantly small number of violation budgets. We employ the neural network ensemble model to estimate the prediction uncertainty and use model predictive control as the basic control framework. We propose the robust cross-entropy method to optimize the control sequence considering the model uncertainty an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.07968","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-10-15T18:19:35Z","cross_cats_sorted":["cs.LG","cs.RO"],"title_canon_sha256":"dbabbfb230744382eb2ccc4932edc51393e017cf92aeb8fcf8efb7d3e96e739e","abstract_canon_sha256":"87b72166e8981af41d1135727ebbce37745becafd2184e2b91c7ddf111bb4e97"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:20:42.377242Z","signature_b64":"D3Ih8c002QosLfHucdvKgjsm8+yWh4sfu8mbqFKciQly71Is4ds4taoKBr+rUjkmvUJLhScnSUkk6WSWd0iVCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"587a75e78c14378dea3622abd8b35cf231566cae17b16104adb409344621fbe5","last_reissued_at":"2026-07-05T02:20:42.376799Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:20:42.376799Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Constrained Model-based Reinforcement Learning with Robust Cross-Entropy Method","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.RO"],"primary_cat":"cs.AI","authors_text":"Baiming Chen, Ding Zhao, Hongyi Zhou, Martial Hebert, Sicheng Zhong, Zuxin Liu","submitted_at":"2020-10-15T18:19:35Z","abstract_excerpt":"This paper studies the constrained/safe reinforcement learning (RL) problem with sparse indicator signals for constraint violations. We propose a model-based approach to enable RL agents to effectively explore the environment with unknown system dynamics and environment constraints given a significantly small number of violation budgets. We employ the neural network ensemble model to estimate the prediction uncertainty and use model predictive control as the basic control framework. We propose the robust cross-entropy method to optimize the control sequence considering the model uncertainty an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.07968","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.07968/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.07968","created_at":"2026-07-05T02:20:42.376858+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.07968v2","created_at":"2026-07-05T02:20:42.376858+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.07968","created_at":"2026-07-05T02:20:42.376858+00:00"},{"alias_kind":"pith_short_12","alias_value":"LB5HLZ4MCQ3Y","created_at":"2026-07-05T02:20:42.376858+00:00"},{"alias_kind":"pith_short_16","alias_value":"LB5HLZ4MCQ3Y32RW","created_at":"2026-07-05T02:20:42.376858+00:00"},{"alias_kind":"pith_short_8","alias_value":"LB5HLZ4M","created_at":"2026-07-05T02:20:42.376858+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.04828","citing_title":"Safe Planning and Policy Optimization via World Model Learning","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LB5HLZ4MCQ3Y32RWEKV5RM246I","json":"https://pith.science/pith/LB5HLZ4MCQ3Y32RWEKV5RM246I.json","graph_json":"https://pith.science/api/pith-number/LB5HLZ4MCQ3Y32RWEKV5RM246I/graph.json","events_json":"https://pith.science/api/pith-number/LB5HLZ4MCQ3Y32RWEKV5RM246I/events.json","paper":"https://pith.science/paper/LB5HLZ4M"},"agent_actions":{"view_html":"https://pith.science/pith/LB5HLZ4MCQ3Y32RWEKV5RM246I","download_json":"https://pith.science/pith/LB5HLZ4MCQ3Y32RWEKV5RM246I.json","view_paper":"https://pith.science/paper/LB5HLZ4M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.07968&json=true","fetch_graph":"https://pith.science/api/pith-number/LB5HLZ4MCQ3Y32RWEKV5RM246I/graph.json","fetch_events":"https://pith.science/api/pith-number/LB5HLZ4MCQ3Y32RWEKV5RM246I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LB5HLZ4MCQ3Y32RWEKV5RM246I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LB5HLZ4MCQ3Y32RWEKV5RM246I/action/storage_attestation","attest_author":"https://pith.science/pith/LB5HLZ4MCQ3Y32RWEKV5RM246I/action/author_attestation","sign_citation":"https://pith.science/pith/LB5HLZ4MCQ3Y32RWEKV5RM246I/action/citation_signature","submit_replication":"https://pith.science/pith/LB5HLZ4MCQ3Y32RWEKV5RM246I/action/replication_record"}},"created_at":"2026-07-05T02:20:42.376858+00:00","updated_at":"2026-07-05T02:20:42.376858+00:00"}