{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:B6N75HTUAXUVSMTXHTKZKJLSUH","short_pith_number":"pith:B6N75HTU","schema_version":"1.0","canonical_sha256":"0f9bfe9e7405e95932773cd5952572a1c5dee8e905fc2892963202760a4f3365","source":{"kind":"arxiv","id":"2201.00762","version":2},"attestation_state":"computed","paper":{"title":"Execute Order 66: Targeted Data Poisoning for Reinforcement Learning","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.CR"],"primary_cat":"cs.LG","authors_text":"Gavin Taylor, Harrison Foley, Liam Fowl, Tom Goldstein","submitted_at":"2022-01-03T17:09:32Z","abstract_excerpt":"Data poisoning for reinforcement learning has historically focused on general performance degradation, and targeted attacks have been successful via perturbations that involve control of the victim's policy and rewards. We introduce an insidious poisoning attack for reinforcement learning which causes agent misbehavior only at specific target states - all while minimally modifying a small fraction of training observations without assuming any control over policy or reward. We accomplish this by adapting a recent technique, gradient alignment, to reinforcement learning. We test our method and d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.00762","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.LG","submitted_at":"2022-01-03T17:09:32Z","cross_cats_sorted":["cs.AI","cs.CR"],"title_canon_sha256":"26301eaa6ca31e34500ac145ad0bd7751e0f367389b3d7637fd2988810d8f8e9","abstract_canon_sha256":"492c72ae67cd2c8a9c387908fdf67f454aa4dd4d8a57f619c3ec8badcd632226"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:44:19.439221Z","signature_b64":"K1FSxWUgh+WU3FGV6pSABEBVWH6alKB/aUc31MtjHxBvDR0LoQM0Foo2Tqvutl6vhARvvKb7FPlxe+sYCbpiDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0f9bfe9e7405e95932773cd5952572a1c5dee8e905fc2892963202760a4f3365","last_reissued_at":"2026-07-05T04:44:19.438733Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:44:19.438733Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Execute Order 66: Targeted Data Poisoning for Reinforcement Learning","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.CR"],"primary_cat":"cs.LG","authors_text":"Gavin Taylor, Harrison Foley, Liam Fowl, Tom Goldstein","submitted_at":"2022-01-03T17:09:32Z","abstract_excerpt":"Data poisoning for reinforcement learning has historically focused on general performance degradation, and targeted attacks have been successful via perturbations that involve control of the victim's policy and rewards. We introduce an insidious poisoning attack for reinforcement learning which causes agent misbehavior only at specific target states - all while minimally modifying a small fraction of training observations without assuming any control over policy or reward. We accomplish this by adapting a recent technique, gradient alignment, to reinforcement learning. We test our method and d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.00762","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.00762/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.00762","created_at":"2026-07-05T04:44:19.438792+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.00762v2","created_at":"2026-07-05T04:44:19.438792+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.00762","created_at":"2026-07-05T04:44:19.438792+00:00"},{"alias_kind":"pith_short_12","alias_value":"B6N75HTUAXUV","created_at":"2026-07-05T04:44:19.438792+00:00"},{"alias_kind":"pith_short_16","alias_value":"B6N75HTUAXUVSMTX","created_at":"2026-07-05T04:44:19.438792+00:00"},{"alias_kind":"pith_short_8","alias_value":"B6N75HTU","created_at":"2026-07-05T04:44:19.438792+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.04883","citing_title":"Beyond Training-time Poisoning: Component-level and Post-training Backdoors in Deep Reinforcement Learning","ref_index":2022,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B6N75HTUAXUVSMTXHTKZKJLSUH","json":"https://pith.science/pith/B6N75HTUAXUVSMTXHTKZKJLSUH.json","graph_json":"https://pith.science/api/pith-number/B6N75HTUAXUVSMTXHTKZKJLSUH/graph.json","events_json":"https://pith.science/api/pith-number/B6N75HTUAXUVSMTXHTKZKJLSUH/events.json","paper":"https://pith.science/paper/B6N75HTU"},"agent_actions":{"view_html":"https://pith.science/pith/B6N75HTUAXUVSMTXHTKZKJLSUH","download_json":"https://pith.science/pith/B6N75HTUAXUVSMTXHTKZKJLSUH.json","view_paper":"https://pith.science/paper/B6N75HTU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.00762&json=true","fetch_graph":"https://pith.science/api/pith-number/B6N75HTUAXUVSMTXHTKZKJLSUH/graph.json","fetch_events":"https://pith.science/api/pith-number/B6N75HTUAXUVSMTXHTKZKJLSUH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B6N75HTUAXUVSMTXHTKZKJLSUH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B6N75HTUAXUVSMTXHTKZKJLSUH/action/storage_attestation","attest_author":"https://pith.science/pith/B6N75HTUAXUVSMTXHTKZKJLSUH/action/author_attestation","sign_citation":"https://pith.science/pith/B6N75HTUAXUVSMTXHTKZKJLSUH/action/citation_signature","submit_replication":"https://pith.science/pith/B6N75HTUAXUVSMTXHTKZKJLSUH/action/replication_record"}},"created_at":"2026-07-05T04:44:19.438792+00:00","updated_at":"2026-07-05T04:44:19.438792+00:00"}