{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7SZZFKAIZLGCVDWD5HUXJXHBUH","short_pith_number":"pith:7SZZFKAI","schema_version":"1.0","canonical_sha256":"fcb392a808cacc2a8ec3e9e974dce1a1c5cdd8639031cd269148330228313cfb","source":{"kind":"arxiv","id":"2405.19909","version":3},"attestation_state":"computed","paper":{"title":"Adaptive Advantage-Guided Policy Regularization for Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Hao Gao, Tenglong Liu, Wei Pan, Xin Xu, Yang Li, Yixing Lan","submitted_at":"2024-05-30T10:20:55Z","abstract_excerpt":"In offline reinforcement learning, the challenge of out-of-distribution (OOD) is pronounced. To address this, existing methods often constrain the learned policy through policy regularization. However, these methods often suffer from the issue of unnecessary conservativeness, hampering policy improvement. This occurs due to the indiscriminate use of all actions from the behavior policy that generates the offline dataset as constraints. The problem becomes particularly noticeable when the quality of the dataset is suboptimal. Thus, we propose Adaptive Advantage-guided Policy Regularization (A2P"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.19909","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-30T10:20:55Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"0e505d3dd243df2a462d889c63760b4436cef41b610305457ca2f429456b9e84","abstract_canon_sha256":"1b0550e071aa06232d1f2b563bcdb05ba8e790fa4e380ffbacb9a6af3cfce208"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:43:56.120750Z","signature_b64":"YGHpphLSrsV2GUgJjAnjlKVNWzjmwn/vVSYN3KaZ2lQAQr9zQz3gGbQJuUlErvlUzMq+ZAL6P+heWFMXUReGAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fcb392a808cacc2a8ec3e9e974dce1a1c5cdd8639031cd269148330228313cfb","last_reissued_at":"2026-07-05T08:43:56.120263Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:43:56.120263Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adaptive Advantage-Guided Policy Regularization for Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Hao Gao, Tenglong Liu, Wei Pan, Xin Xu, Yang Li, Yixing Lan","submitted_at":"2024-05-30T10:20:55Z","abstract_excerpt":"In offline reinforcement learning, the challenge of out-of-distribution (OOD) is pronounced. To address this, existing methods often constrain the learned policy through policy regularization. However, these methods often suffer from the issue of unnecessary conservativeness, hampering policy improvement. This occurs due to the indiscriminate use of all actions from the behavior policy that generates the offline dataset as constraints. The problem becomes particularly noticeable when the quality of the dataset is suboptimal. Thus, we propose Adaptive Advantage-guided Policy Regularization (A2P"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.19909","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.19909/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.19909","created_at":"2026-07-05T08:43:56.120322+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.19909v3","created_at":"2026-07-05T08:43:56.120322+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.19909","created_at":"2026-07-05T08:43:56.120322+00:00"},{"alias_kind":"pith_short_12","alias_value":"7SZZFKAIZLGC","created_at":"2026-07-05T08:43:56.120322+00:00"},{"alias_kind":"pith_short_16","alias_value":"7SZZFKAIZLGCVDWD","created_at":"2026-07-05T08:43:56.120322+00:00"},{"alias_kind":"pith_short_8","alias_value":"7SZZFKAI","created_at":"2026-07-05T08:43:56.120322+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.08202","citing_title":"Beyond Penalization: Diffusion-based Out-of-Distribution Detection and Selective Regularization in Offline Reinforcement Learning","ref_index":57,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7SZZFKAIZLGCVDWD5HUXJXHBUH","json":"https://pith.science/pith/7SZZFKAIZLGCVDWD5HUXJXHBUH.json","graph_json":"https://pith.science/api/pith-number/7SZZFKAIZLGCVDWD5HUXJXHBUH/graph.json","events_json":"https://pith.science/api/pith-number/7SZZFKAIZLGCVDWD5HUXJXHBUH/events.json","paper":"https://pith.science/paper/7SZZFKAI"},"agent_actions":{"view_html":"https://pith.science/pith/7SZZFKAIZLGCVDWD5HUXJXHBUH","download_json":"https://pith.science/pith/7SZZFKAIZLGCVDWD5HUXJXHBUH.json","view_paper":"https://pith.science/paper/7SZZFKAI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.19909&json=true","fetch_graph":"https://pith.science/api/pith-number/7SZZFKAIZLGCVDWD5HUXJXHBUH/graph.json","fetch_events":"https://pith.science/api/pith-number/7SZZFKAIZLGCVDWD5HUXJXHBUH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7SZZFKAIZLGCVDWD5HUXJXHBUH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7SZZFKAIZLGCVDWD5HUXJXHBUH/action/storage_attestation","attest_author":"https://pith.science/pith/7SZZFKAIZLGCVDWD5HUXJXHBUH/action/author_attestation","sign_citation":"https://pith.science/pith/7SZZFKAIZLGCVDWD5HUXJXHBUH/action/citation_signature","submit_replication":"https://pith.science/pith/7SZZFKAIZLGCVDWD5HUXJXHBUH/action/replication_record"}},"created_at":"2026-07-05T08:43:56.120322+00:00","updated_at":"2026-07-05T08:43:56.120322+00:00"}