{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:UILXP2VL24IC6EPKCGAOAUFEXN","short_pith_number":"pith:UILXP2VL","canonical_record":{"source":{"id":"2109.14678","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-09-29T19:29:10Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b8fea08a0aac33b889b1e3bec56c5f139c8ff33d1f63a4745a6438934fd875b6","abstract_canon_sha256":"a83b4cf43c8c647431effc196cf57bc4a2982c94cac2a23efdcac8a96d3fdda6"},"schema_version":"1.0"},"canonical_sha256":"a21777eaabd7102f11ea1180e050a4bb5ba2f640ad6dd3a2f5fe6b79e4ffcc95","source":{"kind":"arxiv","id":"2109.14678","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2109.14678","created_at":"2026-07-05T03:18:43Z"},{"alias_kind":"arxiv_version","alias_value":"2109.14678v1","created_at":"2026-07-05T03:18:43Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.14678","created_at":"2026-07-05T03:18:43Z"},{"alias_kind":"pith_short_12","alias_value":"UILXP2VL24IC","created_at":"2026-07-05T03:18:43Z"},{"alias_kind":"pith_short_16","alias_value":"UILXP2VL24IC6EPK","created_at":"2026-07-05T03:18:43Z"},{"alias_kind":"pith_short_8","alias_value":"UILXP2VL","created_at":"2026-07-05T03:18:43Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:UILXP2VL24IC6EPKCGAOAUFEXN","target":"record","payload":{"canonical_record":{"source":{"id":"2109.14678","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-09-29T19:29:10Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b8fea08a0aac33b889b1e3bec56c5f139c8ff33d1f63a4745a6438934fd875b6","abstract_canon_sha256":"a83b4cf43c8c647431effc196cf57bc4a2982c94cac2a23efdcac8a96d3fdda6"},"schema_version":"1.0"},"canonical_sha256":"a21777eaabd7102f11ea1180e050a4bb5ba2f640ad6dd3a2f5fe6b79e4ffcc95","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:18:43.268879Z","signature_b64":"OZ3i1sVPRiV877nn+R6qZMdC+zmutJH+9U9jNaB1rX5tEqaxMznfvZT0wlrS2kllNzFH93aJpXV+4Og3sKSeAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a21777eaabd7102f11ea1180e050a4bb5ba2f640ad6dd3a2f5fe6b79e4ffcc95","last_reissued_at":"2026-07-05T03:18:43.268286Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:18:43.268286Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2109.14678","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:18:43Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"8LNaTP9b2pzxryY/ZPW6FpcJ0CF20FRPgRvLiNWgt1sg7XJjSl6v3erjNfB9VTTVacdPM+soQzkYAS21Pba2Dw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-24T03:07:36.227870Z"},"content_sha256":"d6208664f6bafbab1bbb12bb5807c87bfff0b8c8a2f4a754adc8cd2ccaed3e95","schema_version":"1.0","event_id":"sha256:d6208664f6bafbab1bbb12bb5807c87bfff0b8c8a2f4a754adc8cd2ccaed3e95"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:UILXP2VL24IC6EPKCGAOAUFEXN","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Mitigation of Adversarial Policy Imitation via Constrained Randomization of Policy (CRoP)","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Nancirose Piazza, Vahid Behzadan","submitted_at":"2021-09-29T19:29:10Z","abstract_excerpt":"Deep reinforcement learning (DRL) policies are vulnerable to unauthorized replication attacks, where an adversary exploits imitation learning to reproduce target policies from observed behavior. In this paper, we propose Constrained Randomization of Policy (CRoP) as a mitigation technique against such attacks. CRoP induces the execution of sub-optimal actions at random under performance loss constraints. We present a parametric analysis of CRoP, address the optimality of CRoP, and establish theoretical bounds on the adversarial budget and the expectation of loss. Furthermore, we report the exp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.14678","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.14678/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:18:43Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"K4jsOdEI0LhMsZvkFTyTEXiQt5mMdZSjUeEBgTeKzf6P667RiBMurFic/YmbCpOdWsmDNgaSObPTd9yrcdOYCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-24T03:07:36.228260Z"},"content_sha256":"17e67a820947a2b270dc0cb7bd5c24ba6571221225bca2cbcec61a663d82357b","schema_version":"1.0","event_id":"sha256:17e67a820947a2b270dc0cb7bd5c24ba6571221225bca2cbcec61a663d82357b"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/UILXP2VL24IC6EPKCGAOAUFEXN/bundle.json","state_url":"https://pith.science/pith/UILXP2VL24IC6EPKCGAOAUFEXN/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/UILXP2VL24IC6EPKCGAOAUFEXN/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-24T03:07:36Z","links":{"resolver":"https://pith.science/pith/UILXP2VL24IC6EPKCGAOAUFEXN","bundle":"https://pith.science/pith/UILXP2VL24IC6EPKCGAOAUFEXN/bundle.json","state":"https://pith.science/pith/UILXP2VL24IC6EPKCGAOAUFEXN/state.json","well_known_bundle":"https://pith.science/.well-known/pith/UILXP2VL24IC6EPKCGAOAUFEXN/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:UILXP2VL24IC6EPKCGAOAUFEXN","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"a83b4cf43c8c647431effc196cf57bc4a2982c94cac2a23efdcac8a96d3fdda6","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-09-29T19:29:10Z","title_canon_sha256":"b8fea08a0aac33b889b1e3bec56c5f139c8ff33d1f63a4745a6438934fd875b6"},"schema_version":"1.0","source":{"id":"2109.14678","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2109.14678","created_at":"2026-07-05T03:18:43Z"},{"alias_kind":"arxiv_version","alias_value":"2109.14678v1","created_at":"2026-07-05T03:18:43Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.14678","created_at":"2026-07-05T03:18:43Z"},{"alias_kind":"pith_short_12","alias_value":"UILXP2VL24IC","created_at":"2026-07-05T03:18:43Z"},{"alias_kind":"pith_short_16","alias_value":"UILXP2VL24IC6EPK","created_at":"2026-07-05T03:18:43Z"},{"alias_kind":"pith_short_8","alias_value":"UILXP2VL","created_at":"2026-07-05T03:18:43Z"}],"graph_snapshots":[{"event_id":"sha256:17e67a820947a2b270dc0cb7bd5c24ba6571221225bca2cbcec61a663d82357b","target":"graph","created_at":"2026-07-05T03:18:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2109.14678/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Deep reinforcement learning (DRL) policies are vulnerable to unauthorized replication attacks, where an adversary exploits imitation learning to reproduce target policies from observed behavior. In this paper, we propose Constrained Randomization of Policy (CRoP) as a mitigation technique against such attacks. CRoP induces the execution of sub-optimal actions at random under performance loss constraints. We present a parametric analysis of CRoP, address the optimality of CRoP, and establish theoretical bounds on the adversarial budget and the expectation of loss. Furthermore, we report the exp","authors_text":"Nancirose Piazza, Vahid Behzadan","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-09-29T19:29:10Z","title":"Mitigation of Adversarial Policy Imitation via Constrained Randomization of Policy (CRoP)"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.14678","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:d6208664f6bafbab1bbb12bb5807c87bfff0b8c8a2f4a754adc8cd2ccaed3e95","target":"record","created_at":"2026-07-05T03:18:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"a83b4cf43c8c647431effc196cf57bc4a2982c94cac2a23efdcac8a96d3fdda6","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-09-29T19:29:10Z","title_canon_sha256":"b8fea08a0aac33b889b1e3bec56c5f139c8ff33d1f63a4745a6438934fd875b6"},"schema_version":"1.0","source":{"id":"2109.14678","kind":"arxiv","version":1}},"canonical_sha256":"a21777eaabd7102f11ea1180e050a4bb5ba2f640ad6dd3a2f5fe6b79e4ffcc95","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"a21777eaabd7102f11ea1180e050a4bb5ba2f640ad6dd3a2f5fe6b79e4ffcc95","first_computed_at":"2026-07-05T03:18:43.268286Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:18:43.268286Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"OZ3i1sVPRiV877nn+R6qZMdC+zmutJH+9U9jNaB1rX5tEqaxMznfvZT0wlrS2kllNzFH93aJpXV+4Og3sKSeAA==","signature_status":"signed_v1","signed_at":"2026-07-05T03:18:43.268879Z","signed_message":"canonical_sha256_bytes"},"source_id":"2109.14678","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:d6208664f6bafbab1bbb12bb5807c87bfff0b8c8a2f4a754adc8cd2ccaed3e95","sha256:17e67a820947a2b270dc0cb7bd5c24ba6571221225bca2cbcec61a663d82357b"],"state_sha256":"117c5aabd7d122b4d4d9f20755b28958aed245b8bcf40bb3b07ab20615526ff1"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"geR2UKGkq19i66seC4DvXXgUCGnbxVoEFZbR+Sy8H/Vs61DATtiNI3ioDXMd6/QUej+k5RIugiO7NX0JJQ3DCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-24T03:07:36.230942Z","bundle_sha256":"6a6d7622e4a748a391358185e768bd33805b5c0d54cff4da5a30e934c4311efc"}}