{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:EPGMEN4BRZZBQDT5RQB3EXFVCK","short_pith_number":"pith:EPGMEN4B","schema_version":"1.0","canonical_sha256":"23ccc237818e72180e7d8c03b25cb512bd00d4283e3461d6f754fb22cc1dc332","source":{"kind":"arxiv","id":"2105.00579","version":3},"attestation_state":"computed","paper":{"title":"BACKDOORL: Backdoor Attack against Competitive Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Dawn Song, Lun Wang, Wenbo Guo, Xian Wu, Xinyu Xing, Zaynah Javed","submitted_at":"2021-05-02T23:47:55Z","abstract_excerpt":"Recent research has confirmed the feasibility of backdoor attacks in deep reinforcement learning (RL) systems. However, the existing attacks require the ability to arbitrarily modify an agent's observation, constraining the application scope to simple RL systems such as Atari games. In this paper, we migrate backdoor attacks to more complex RL systems involving multiple agents and explore the possibility of triggering the backdoor without directly manipulating the agent's observation. As a proof of concept, we demonstrate that an adversary agent can trigger the backdoor of the victim agent wit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.00579","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2021-05-02T23:47:55Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"f8558fb83a17aa9f8d2280d2383560c812b622a1c18a402536be0a84e6fbe3e9","abstract_canon_sha256":"f3fb61b1b48c2e12db3bd60a62142089f5aa35599b40c478d77976df3630704d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:39:58.890722Z","signature_b64":"bgaz8KEdGrNXr0oP5uvsOle7JuP2nTcRO0iciILPMiItRAhlUDqm3ejc+pKfKy8E2gSFFHaiFwo6yXq9+qx5Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"23ccc237818e72180e7d8c03b25cb512bd00d4283e3461d6f754fb22cc1dc332","last_reissued_at":"2026-07-05T03:39:58.890209Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:39:58.890209Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BACKDOORL: Backdoor Attack against Competitive Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Dawn Song, Lun Wang, Wenbo Guo, Xian Wu, Xinyu Xing, Zaynah Javed","submitted_at":"2021-05-02T23:47:55Z","abstract_excerpt":"Recent research has confirmed the feasibility of backdoor attacks in deep reinforcement learning (RL) systems. However, the existing attacks require the ability to arbitrarily modify an agent's observation, constraining the application scope to simple RL systems such as Atari games. In this paper, we migrate backdoor attacks to more complex RL systems involving multiple agents and explore the possibility of triggering the backdoor without directly manipulating the agent's observation. As a proof of concept, we demonstrate that an adversary agent can trigger the backdoor of the victim agent wit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.00579","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.00579/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.00579","created_at":"2026-07-05T03:39:58.890268+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.00579v3","created_at":"2026-07-05T03:39:58.890268+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.00579","created_at":"2026-07-05T03:39:58.890268+00:00"},{"alias_kind":"pith_short_12","alias_value":"EPGMEN4BRZZB","created_at":"2026-07-05T03:39:58.890268+00:00"},{"alias_kind":"pith_short_16","alias_value":"EPGMEN4BRZZBQDT5","created_at":"2026-07-05T03:39:58.890268+00:00"},{"alias_kind":"pith_short_8","alias_value":"EPGMEN4B","created_at":"2026-07-05T03:39:58.890268+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12896","citing_title":"PolicyGuard: Towards Test-time and Step-level Adversary (Backdoor) Defense for Reinforcement Learning Agent","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00414","citing_title":"Auditing Near-Optimal Policies Can Be Exponentially Hard: Conditional Query Lower Bounds via Occupancy Rashomon Capacity","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05977","citing_title":"BehaviorGuard: Online Backdoor Defense for Deep Reinforcement Learning","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EPGMEN4BRZZBQDT5RQB3EXFVCK","json":"https://pith.science/pith/EPGMEN4BRZZBQDT5RQB3EXFVCK.json","graph_json":"https://pith.science/api/pith-number/EPGMEN4BRZZBQDT5RQB3EXFVCK/graph.json","events_json":"https://pith.science/api/pith-number/EPGMEN4BRZZBQDT5RQB3EXFVCK/events.json","paper":"https://pith.science/paper/EPGMEN4B"},"agent_actions":{"view_html":"https://pith.science/pith/EPGMEN4BRZZBQDT5RQB3EXFVCK","download_json":"https://pith.science/pith/EPGMEN4BRZZBQDT5RQB3EXFVCK.json","view_paper":"https://pith.science/paper/EPGMEN4B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.00579&json=true","fetch_graph":"https://pith.science/api/pith-number/EPGMEN4BRZZBQDT5RQB3EXFVCK/graph.json","fetch_events":"https://pith.science/api/pith-number/EPGMEN4BRZZBQDT5RQB3EXFVCK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EPGMEN4BRZZBQDT5RQB3EXFVCK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EPGMEN4BRZZBQDT5RQB3EXFVCK/action/storage_attestation","attest_author":"https://pith.science/pith/EPGMEN4BRZZBQDT5RQB3EXFVCK/action/author_attestation","sign_citation":"https://pith.science/pith/EPGMEN4BRZZBQDT5RQB3EXFVCK/action/citation_signature","submit_replication":"https://pith.science/pith/EPGMEN4BRZZBQDT5RQB3EXFVCK/action/replication_record"}},"created_at":"2026-07-05T03:39:58.890268+00:00","updated_at":"2026-07-05T03:39:58.890268+00:00"}