{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:X2KGUTDOGHO4VBCTTRKOVPIZBD","short_pith_number":"pith:X2KGUTDO","schema_version":"1.0","canonical_sha256":"be946a4c6e31ddca84539c54eabd1908e9abeba6fc9663b70ab7184980d6e3e0","source":{"kind":"arxiv","id":"2010.14497","version":2},"attestation_state":"computed","paper":{"title":"Conservative Safety Critics for Exploration","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Animesh Garg, Aviral Kumar, Florian Shkurti, Homanga Bharadhwaj, Nicholas Rhinehart, Sergey Levine","submitted_at":"2020-10-27T17:54:25Z","abstract_excerpt":"Safe exploration presents a major challenge in reinforcement learning (RL): when active data collection requires deploying partially trained policies, we must ensure that these policies avoid catastrophically unsafe regions, while still enabling trial and error learning. In this paper, we target the problem of safe exploration in RL by learning a conservative safety estimate of environment states through a critic, and provably upper bound the likelihood of catastrophic failures at every training iteration. We theoretically characterize the tradeoff between safety and policy improvement, show t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.14497","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-10-27T17:54:25Z","cross_cats_sorted":["cs.AI","cs.RO","stat.ML"],"title_canon_sha256":"2b2b577d0850fd5f27edf10d450f1bfe57ffe4fcf9e7692d8451dba38bf2030e","abstract_canon_sha256":"87bb73504ef86487018e3ca98f3675ef7722d90bfed45bf2a3d3cc1daa384b2a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:34:42.898827Z","signature_b64":"OI47hgil47mhwlyW5OQ5ACE6J18YTv0Hz1IhynQdA6MVVaPaSY7CLg3GkTeBGCaWmR61m6imVWX3pVGemJUKDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"be946a4c6e31ddca84539c54eabd1908e9abeba6fc9663b70ab7184980d6e3e0","last_reissued_at":"2026-07-05T02:34:42.898381Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:34:42.898381Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Conservative Safety Critics for Exploration","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Animesh Garg, Aviral Kumar, Florian Shkurti, Homanga Bharadhwaj, Nicholas Rhinehart, Sergey Levine","submitted_at":"2020-10-27T17:54:25Z","abstract_excerpt":"Safe exploration presents a major challenge in reinforcement learning (RL): when active data collection requires deploying partially trained policies, we must ensure that these policies avoid catastrophically unsafe regions, while still enabling trial and error learning. In this paper, we target the problem of safe exploration in RL by learning a conservative safety estimate of environment states through a critic, and provably upper bound the likelihood of catastrophic failures at every training iteration. We theoretically characterize the tradeoff between safety and policy improvement, show t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.14497","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.14497/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.14497","created_at":"2026-07-05T02:34:42.898433+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.14497v2","created_at":"2026-07-05T02:34:42.898433+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.14497","created_at":"2026-07-05T02:34:42.898433+00:00"},{"alias_kind":"pith_short_12","alias_value":"X2KGUTDOGHO4","created_at":"2026-07-05T02:34:42.898433+00:00"},{"alias_kind":"pith_short_16","alias_value":"X2KGUTDOGHO4VBCT","created_at":"2026-07-05T02:34:42.898433+00:00"},{"alias_kind":"pith_short_8","alias_value":"X2KGUTDO","created_at":"2026-07-05T02:34:42.898433+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12372","citing_title":"UniIntervene: Agentic Intervention for Efficient Real-World Reinforcement Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10228","citing_title":"SHAPO: Sharpness-Aware Policy Optimization for Safe Exploration","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05660","citing_title":"Safe Embodied AI for Long-horizon Tasks: A Cross-layer Analysis of Robotic Manipulation","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09825","citing_title":"An Agency-Transferring Model-Free Policy Enhancement Technique","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22446","citing_title":"Pre-VLA: Preemptive Runtime Verification for Reliable Vision-Language-Action and World-Model Rollouts","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25379","citing_title":"Safe-Support Q-Learning: Learning without Unsafe Exploration","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X2KGUTDOGHO4VBCTTRKOVPIZBD","json":"https://pith.science/pith/X2KGUTDOGHO4VBCTTRKOVPIZBD.json","graph_json":"https://pith.science/api/pith-number/X2KGUTDOGHO4VBCTTRKOVPIZBD/graph.json","events_json":"https://pith.science/api/pith-number/X2KGUTDOGHO4VBCTTRKOVPIZBD/events.json","paper":"https://pith.science/paper/X2KGUTDO"},"agent_actions":{"view_html":"https://pith.science/pith/X2KGUTDOGHO4VBCTTRKOVPIZBD","download_json":"https://pith.science/pith/X2KGUTDOGHO4VBCTTRKOVPIZBD.json","view_paper":"https://pith.science/paper/X2KGUTDO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.14497&json=true","fetch_graph":"https://pith.science/api/pith-number/X2KGUTDOGHO4VBCTTRKOVPIZBD/graph.json","fetch_events":"https://pith.science/api/pith-number/X2KGUTDOGHO4VBCTTRKOVPIZBD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X2KGUTDOGHO4VBCTTRKOVPIZBD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X2KGUTDOGHO4VBCTTRKOVPIZBD/action/storage_attestation","attest_author":"https://pith.science/pith/X2KGUTDOGHO4VBCTTRKOVPIZBD/action/author_attestation","sign_citation":"https://pith.science/pith/X2KGUTDOGHO4VBCTTRKOVPIZBD/action/citation_signature","submit_replication":"https://pith.science/pith/X2KGUTDOGHO4VBCTTRKOVPIZBD/action/replication_record"}},"created_at":"2026-07-05T02:34:42.898433+00:00","updated_at":"2026-07-05T02:34:42.898433+00:00"}