{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:D5FPSTRLPXQNR2GLF5UCR44XHP","short_pith_number":"pith:D5FPSTRL","schema_version":"1.0","canonical_sha256":"1f4af94e2b7de0d8e8cb2f6828f3973bec3a95cd4d2a873f759976bd9c0c453b","source":{"kind":"arxiv","id":"2501.05501","version":2},"attestation_state":"computed","paper":{"title":"Strategy Masking: A Method for Guardrails in Value-based Reinforcement Learning Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.MA"],"primary_cat":"cs.AI","authors_text":"Jeremy Kedziora, Jonathan Keane, Sam Keyser","submitted_at":"2025-01-09T18:43:05Z","abstract_excerpt":"The use of reward functions to structure AI learning and decision making is core to the current reinforcement learning paradigm; however, without careful design of reward functions, agents can learn to solve problems in ways that may be considered \"undesirable\" or \"unethical.\" Without thorough understanding of the incentives a reward function creates, it can be difficult to impose principled yet general control mechanisms over its behavior. In this paper, we study methods for constructing guardrails for AI agents that use reward functions to learn decision making. We introduce a novel approach"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.05501","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-01-09T18:43:05Z","cross_cats_sorted":["cs.LG","cs.MA"],"title_canon_sha256":"8a459c1f734e93fb52b19146eaf95e0925d8a5caace40104954e7aab5d357454","abstract_canon_sha256":"20484a7c46319f9434fd032d8ca9c631ea2c7d15eb4c72b12cd0fd8f2fdfbea9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:02:57.679881Z","signature_b64":"bzXfH6P+SxzjCOio9EsxjyE8J8buAXtpT+UcNYKoRODqYxGpKkz2IZUH5icIOIH+Ke9/hpQqBevTRrA5dEmEAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1f4af94e2b7de0d8e8cb2f6828f3973bec3a95cd4d2a873f759976bd9c0c453b","last_reissued_at":"2026-07-05T10:02:57.679325Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:02:57.679325Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Strategy Masking: A Method for Guardrails in Value-based Reinforcement Learning Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.MA"],"primary_cat":"cs.AI","authors_text":"Jeremy Kedziora, Jonathan Keane, Sam Keyser","submitted_at":"2025-01-09T18:43:05Z","abstract_excerpt":"The use of reward functions to structure AI learning and decision making is core to the current reinforcement learning paradigm; however, without careful design of reward functions, agents can learn to solve problems in ways that may be considered \"undesirable\" or \"unethical.\" Without thorough understanding of the incentives a reward function creates, it can be difficult to impose principled yet general control mechanisms over its behavior. In this paper, we study methods for constructing guardrails for AI agents that use reward functions to learn decision making. We introduce a novel approach"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.05501","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.05501/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.05501","created_at":"2026-07-05T10:02:57.679393+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.05501v2","created_at":"2026-07-05T10:02:57.679393+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.05501","created_at":"2026-07-05T10:02:57.679393+00:00"},{"alias_kind":"pith_short_12","alias_value":"D5FPSTRLPXQN","created_at":"2026-07-05T10:02:57.679393+00:00"},{"alias_kind":"pith_short_16","alias_value":"D5FPSTRLPXQNR2GL","created_at":"2026-07-05T10:02:57.679393+00:00"},{"alias_kind":"pith_short_8","alias_value":"D5FPSTRL","created_at":"2026-07-05T10:02:57.679393+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D5FPSTRLPXQNR2GLF5UCR44XHP","json":"https://pith.science/pith/D5FPSTRLPXQNR2GLF5UCR44XHP.json","graph_json":"https://pith.science/api/pith-number/D5FPSTRLPXQNR2GLF5UCR44XHP/graph.json","events_json":"https://pith.science/api/pith-number/D5FPSTRLPXQNR2GLF5UCR44XHP/events.json","paper":"https://pith.science/paper/D5FPSTRL"},"agent_actions":{"view_html":"https://pith.science/pith/D5FPSTRLPXQNR2GLF5UCR44XHP","download_json":"https://pith.science/pith/D5FPSTRLPXQNR2GLF5UCR44XHP.json","view_paper":"https://pith.science/paper/D5FPSTRL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.05501&json=true","fetch_graph":"https://pith.science/api/pith-number/D5FPSTRLPXQNR2GLF5UCR44XHP/graph.json","fetch_events":"https://pith.science/api/pith-number/D5FPSTRLPXQNR2GLF5UCR44XHP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D5FPSTRLPXQNR2GLF5UCR44XHP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D5FPSTRLPXQNR2GLF5UCR44XHP/action/storage_attestation","attest_author":"https://pith.science/pith/D5FPSTRLPXQNR2GLF5UCR44XHP/action/author_attestation","sign_citation":"https://pith.science/pith/D5FPSTRLPXQNR2GLF5UCR44XHP/action/citation_signature","submit_replication":"https://pith.science/pith/D5FPSTRLPXQNR2GLF5UCR44XHP/action/replication_record"}},"created_at":"2026-07-05T10:02:57.679393+00:00","updated_at":"2026-07-05T10:02:57.679393+00:00"}