{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2GAWCXIA3R43DUSNP3ZOXHJSRO","short_pith_number":"pith:2GAWCXIA","schema_version":"1.0","canonical_sha256":"d181615d00dc79b1d24d7ef2eb9d328baa98d34adecc79faa9804e4b7fbe10af","source":{"kind":"arxiv","id":"2501.03562","version":2},"attestation_state":"computed","paper":{"title":"Rethinking Adversarial Attacks in Reinforcement Learning from Policy Distribution Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dong Huang, Heming Cui, Hongbin Liang, Ling Xiong, Tianyang Duan, Xianhao Chen, Yong Cui, Yue Gao, Zheng Lin, Zongyuan Zhang","submitted_at":"2025-01-07T06:22:55Z","abstract_excerpt":"Deep Reinforcement Learning (DRL) suffers from uncertainties and inaccuracies in the observation signal in realworld applications. Adversarial attack is an effective method for evaluating the robustness of DRL agents. However, existing attack methods targeting individual sampled actions have limited impacts on the overall policy distribution, particularly in continuous action spaces. To address these limitations, we propose the Distribution-Aware Projected Gradient Descent attack (DAPGD). DAPGD uses distribution similarity as the gradient perturbation input to attack the policy network, which "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.03562","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-01-07T06:22:55Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1275b2e5863b78355f3dabfbad83f3849bcd1b9b6def550c689598abd8bf8c1d","abstract_canon_sha256":"752a5af872b3c993cd9ae6ab3d6a732c5c75952d72d75c45a0cd59226cc971e1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:58:28.313281Z","signature_b64":"7ufrAOe7BXa8WK2avqE2EB8Af8Np8QVzw5jpJigEwWfx37Tw3WT+KrXg+iKo774JkHDTR2xApj0F7R1Ghj/IDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d181615d00dc79b1d24d7ef2eb9d328baa98d34adecc79faa9804e4b7fbe10af","last_reissued_at":"2026-07-05T09:58:28.312674Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:58:28.312674Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Rethinking Adversarial Attacks in Reinforcement Learning from Policy Distribution Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dong Huang, Heming Cui, Hongbin Liang, Ling Xiong, Tianyang Duan, Xianhao Chen, Yong Cui, Yue Gao, Zheng Lin, Zongyuan Zhang","submitted_at":"2025-01-07T06:22:55Z","abstract_excerpt":"Deep Reinforcement Learning (DRL) suffers from uncertainties and inaccuracies in the observation signal in realworld applications. Adversarial attack is an effective method for evaluating the robustness of DRL agents. However, existing attack methods targeting individual sampled actions have limited impacts on the overall policy distribution, particularly in continuous action spaces. To address these limitations, we propose the Distribution-Aware Projected Gradient Descent attack (DAPGD). DAPGD uses distribution similarity as the gradient perturbation input to attack the policy network, which "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.03562","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.03562/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.03562","created_at":"2026-07-05T09:58:28.312763+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.03562v2","created_at":"2026-07-05T09:58:28.312763+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.03562","created_at":"2026-07-05T09:58:28.312763+00:00"},{"alias_kind":"pith_short_12","alias_value":"2GAWCXIA3R43","created_at":"2026-07-05T09:58:28.312763+00:00"},{"alias_kind":"pith_short_16","alias_value":"2GAWCXIA3R43DUSN","created_at":"2026-07-05T09:58:28.312763+00:00"},{"alias_kind":"pith_short_8","alias_value":"2GAWCXIA","created_at":"2026-07-05T09:58:28.312763+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.10685","citing_title":"Defensive Adversarial CAPTCHA: A Semantics-Driven Framework for Natural Adversarial Example Generation","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2GAWCXIA3R43DUSNP3ZOXHJSRO","json":"https://pith.science/pith/2GAWCXIA3R43DUSNP3ZOXHJSRO.json","graph_json":"https://pith.science/api/pith-number/2GAWCXIA3R43DUSNP3ZOXHJSRO/graph.json","events_json":"https://pith.science/api/pith-number/2GAWCXIA3R43DUSNP3ZOXHJSRO/events.json","paper":"https://pith.science/paper/2GAWCXIA"},"agent_actions":{"view_html":"https://pith.science/pith/2GAWCXIA3R43DUSNP3ZOXHJSRO","download_json":"https://pith.science/pith/2GAWCXIA3R43DUSNP3ZOXHJSRO.json","view_paper":"https://pith.science/paper/2GAWCXIA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.03562&json=true","fetch_graph":"https://pith.science/api/pith-number/2GAWCXIA3R43DUSNP3ZOXHJSRO/graph.json","fetch_events":"https://pith.science/api/pith-number/2GAWCXIA3R43DUSNP3ZOXHJSRO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2GAWCXIA3R43DUSNP3ZOXHJSRO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2GAWCXIA3R43DUSNP3ZOXHJSRO/action/storage_attestation","attest_author":"https://pith.science/pith/2GAWCXIA3R43DUSNP3ZOXHJSRO/action/author_attestation","sign_citation":"https://pith.science/pith/2GAWCXIA3R43DUSNP3ZOXHJSRO/action/citation_signature","submit_replication":"https://pith.science/pith/2GAWCXIA3R43DUSNP3ZOXHJSRO/action/replication_record"}},"created_at":"2026-07-05T09:58:28.312763+00:00","updated_at":"2026-07-05T09:58:28.312763+00:00"}