{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:HJVC4GHF2SW2OAEH57JLRHI6SF","short_pith_number":"pith:HJVC4GHF","schema_version":"1.0","canonical_sha256":"3a6a2e18e5d4ada70087efd2b89d1e915e7c9a0f20bf243779f0c0d459cdadfa","source":{"kind":"arxiv","id":"2106.15587","version":2},"attestation_state":"computed","paper":{"title":"Generalization of Reinforcement Learning with Policy-Aware Adversarial Data Augmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Hanping Zhang, Yuhong Guo","submitted_at":"2021-06-29T17:21:59Z","abstract_excerpt":"The generalization gap in reinforcement learning (RL) has been a significant obstacle that prevents the RL agent from learning general skills and adapting to varying environments. Increasing the generalization capacity of the RL systems can significantly improve their performance on real-world working environments. In this work, we propose a novel policy-aware adversarial data augmentation method to augment the standard policy learning method with automatically generated trajectory data. Different from the commonly used observation transformation based data augmentations, our proposed method a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.15587","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-29T17:21:59Z","cross_cats_sorted":[],"title_canon_sha256":"81ff0f0d1e0c637a0fb2cb2aeff2b4f6f03ceb684bb0610abd12fe187f2160ae","abstract_canon_sha256":"ebcb3b18853ebb7a5b0886eefc5efb03753f3b52a7b6f17da6f17ceb99a5e3a1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:37:26.833366Z","signature_b64":"P9puSppVe6zkVu64y2+SNuxYOeuPWcMVL5R9ZnfbinaqBjmAKSnkJrPbLC8BhbK6g4QItUOT10HNKFLCoZOcDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3a6a2e18e5d4ada70087efd2b89d1e915e7c9a0f20bf243779f0c0d459cdadfa","last_reissued_at":"2026-07-05T03:37:26.832894Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:37:26.832894Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generalization of Reinforcement Learning with Policy-Aware Adversarial Data Augmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Hanping Zhang, Yuhong Guo","submitted_at":"2021-06-29T17:21:59Z","abstract_excerpt":"The generalization gap in reinforcement learning (RL) has been a significant obstacle that prevents the RL agent from learning general skills and adapting to varying environments. Increasing the generalization capacity of the RL systems can significantly improve their performance on real-world working environments. In this work, we propose a novel policy-aware adversarial data augmentation method to augment the standard policy learning method with automatically generated trajectory data. Different from the commonly used observation transformation based data augmentations, our proposed method a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.15587","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.15587/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.15587","created_at":"2026-07-05T03:37:26.832949+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.15587v2","created_at":"2026-07-05T03:37:26.832949+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.15587","created_at":"2026-07-05T03:37:26.832949+00:00"},{"alias_kind":"pith_short_12","alias_value":"HJVC4GHF2SW2","created_at":"2026-07-05T03:37:26.832949+00:00"},{"alias_kind":"pith_short_16","alias_value":"HJVC4GHF2SW2OAEH","created_at":"2026-07-05T03:37:26.832949+00:00"},{"alias_kind":"pith_short_8","alias_value":"HJVC4GHF","created_at":"2026-07-05T03:37:26.832949+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.00347","citing_title":"LLM-Driven Policy Diffusion: Enhancing Generalization in Offline Reinforcement Learning","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HJVC4GHF2SW2OAEH57JLRHI6SF","json":"https://pith.science/pith/HJVC4GHF2SW2OAEH57JLRHI6SF.json","graph_json":"https://pith.science/api/pith-number/HJVC4GHF2SW2OAEH57JLRHI6SF/graph.json","events_json":"https://pith.science/api/pith-number/HJVC4GHF2SW2OAEH57JLRHI6SF/events.json","paper":"https://pith.science/paper/HJVC4GHF"},"agent_actions":{"view_html":"https://pith.science/pith/HJVC4GHF2SW2OAEH57JLRHI6SF","download_json":"https://pith.science/pith/HJVC4GHF2SW2OAEH57JLRHI6SF.json","view_paper":"https://pith.science/paper/HJVC4GHF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.15587&json=true","fetch_graph":"https://pith.science/api/pith-number/HJVC4GHF2SW2OAEH57JLRHI6SF/graph.json","fetch_events":"https://pith.science/api/pith-number/HJVC4GHF2SW2OAEH57JLRHI6SF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HJVC4GHF2SW2OAEH57JLRHI6SF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HJVC4GHF2SW2OAEH57JLRHI6SF/action/storage_attestation","attest_author":"https://pith.science/pith/HJVC4GHF2SW2OAEH57JLRHI6SF/action/author_attestation","sign_citation":"https://pith.science/pith/HJVC4GHF2SW2OAEH57JLRHI6SF/action/citation_signature","submit_replication":"https://pith.science/pith/HJVC4GHF2SW2OAEH57JLRHI6SF/action/replication_record"}},"created_at":"2026-07-05T03:37:26.832949+00:00","updated_at":"2026-07-05T03:37:26.832949+00:00"}