{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:RIVUP7CWH5VM6MOG2QFKUQIYY6","short_pith_number":"pith:RIVUP7CW","schema_version":"1.0","canonical_sha256":"8a2b47fc563f6acf31c6d40aaa4118c7bd9c89ff103fd8aa2e290e511594b97a","source":{"kind":"arxiv","id":"2107.06686","version":1},"attestation_state":"computed","paper":{"title":"Safer Reinforcement Learning through Transferable Instinct Networks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.NE"],"primary_cat":"cs.LG","authors_text":"Djordje Grbic, Sebastian Risi","submitted_at":"2021-07-14T13:22:04Z","abstract_excerpt":"Random exploration is one of the main mechanisms through which reinforcement learning (RL) finds well-performing policies. However, it can lead to undesirable or catastrophic outcomes when learning online in safety-critical environments. In fact, safe learning is one of the major obstacles towards real-world agents that can learn during deployment. One way of ensuring that agents respect hard limitations is to explicitly configure boundaries in which they can operate. While this might work in some cases, we do not always have clear a-priori information which states and actions can lead dangero"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2107.06686","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-07-14T13:22:04Z","cross_cats_sorted":["cs.AI","cs.NE"],"title_canon_sha256":"cb74823bcb78228548537aa24edfe3dde9737a70135709cb36c8eb6bf7c56504","abstract_canon_sha256":"62801f6c717ede87ba6af4c561211f5a225a2a67419014cfeb03a589a1ced91e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:58:00.381161Z","signature_b64":"s5tfOgWc7TlzSi1ce5Bo9QpEt9BYeT2FdzbOtx5P1sINIjALf2/k7GFVENCYZjx3IiqizX/mbi+5dhzMwlalAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8a2b47fc563f6acf31c6d40aaa4118c7bd9c89ff103fd8aa2e290e511594b97a","last_reissued_at":"2026-07-05T02:58:00.380794Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:58:00.380794Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safer Reinforcement Learning through Transferable Instinct Networks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.NE"],"primary_cat":"cs.LG","authors_text":"Djordje Grbic, Sebastian Risi","submitted_at":"2021-07-14T13:22:04Z","abstract_excerpt":"Random exploration is one of the main mechanisms through which reinforcement learning (RL) finds well-performing policies. However, it can lead to undesirable or catastrophic outcomes when learning online in safety-critical environments. In fact, safe learning is one of the major obstacles towards real-world agents that can learn during deployment. One way of ensuring that agents respect hard limitations is to explicitly configure boundaries in which they can operate. While this might work in some cases, we do not always have clear a-priori information which states and actions can lead dangero"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.06686","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.06686/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2107.06686","created_at":"2026-07-05T02:58:00.380843+00:00"},{"alias_kind":"arxiv_version","alias_value":"2107.06686v1","created_at":"2026-07-05T02:58:00.380843+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.06686","created_at":"2026-07-05T02:58:00.380843+00:00"},{"alias_kind":"pith_short_12","alias_value":"RIVUP7CWH5VM","created_at":"2026-07-05T02:58:00.380843+00:00"},{"alias_kind":"pith_short_16","alias_value":"RIVUP7CWH5VM6MOG","created_at":"2026-07-05T02:58:00.380843+00:00"},{"alias_kind":"pith_short_8","alias_value":"RIVUP7CW","created_at":"2026-07-05T02:58:00.380843+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RIVUP7CWH5VM6MOG2QFKUQIYY6","json":"https://pith.science/pith/RIVUP7CWH5VM6MOG2QFKUQIYY6.json","graph_json":"https://pith.science/api/pith-number/RIVUP7CWH5VM6MOG2QFKUQIYY6/graph.json","events_json":"https://pith.science/api/pith-number/RIVUP7CWH5VM6MOG2QFKUQIYY6/events.json","paper":"https://pith.science/paper/RIVUP7CW"},"agent_actions":{"view_html":"https://pith.science/pith/RIVUP7CWH5VM6MOG2QFKUQIYY6","download_json":"https://pith.science/pith/RIVUP7CWH5VM6MOG2QFKUQIYY6.json","view_paper":"https://pith.science/paper/RIVUP7CW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2107.06686&json=true","fetch_graph":"https://pith.science/api/pith-number/RIVUP7CWH5VM6MOG2QFKUQIYY6/graph.json","fetch_events":"https://pith.science/api/pith-number/RIVUP7CWH5VM6MOG2QFKUQIYY6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RIVUP7CWH5VM6MOG2QFKUQIYY6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RIVUP7CWH5VM6MOG2QFKUQIYY6/action/storage_attestation","attest_author":"https://pith.science/pith/RIVUP7CWH5VM6MOG2QFKUQIYY6/action/author_attestation","sign_citation":"https://pith.science/pith/RIVUP7CWH5VM6MOG2QFKUQIYY6/action/citation_signature","submit_replication":"https://pith.science/pith/RIVUP7CWH5VM6MOG2QFKUQIYY6/action/replication_record"}},"created_at":"2026-07-05T02:58:00.380843+00:00","updated_at":"2026-07-05T02:58:00.380843+00:00"}