{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:66JHKFWKZ6TCWXUSLJPWJDVGNN","short_pith_number":"pith:66JHKFWK","schema_version":"1.0","canonical_sha256":"f7927516cacfa62b5e925a5f648ea66b5c28451891a5d09ff99c13971b5c07d1","source":{"kind":"arxiv","id":"2203.06662","version":1},"attestation_state":"computed","paper":{"title":"DARA: Dynamics-Aware Reward Augmentation in Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Donglin Wang, Hongyin Zhang, Jinxin Liu","submitted_at":"2022-03-13T14:30:55Z","abstract_excerpt":"Offline reinforcement learning algorithms promise to be applicable in settings where a fixed dataset is available and no new experience can be acquired. However, such formulation is inevitably offline-data-hungry and, in practice, collecting a large offline dataset for one specific task over one specific environment is also costly and laborious. In this paper, we thus 1) formulate the offline dynamics adaptation by using (source) offline data collected from another dynamics to relax the requirement for the extensive (target) offline data, 2) characterize the dynamics shift problem in which pri"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.06662","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-03-13T14:30:55Z","cross_cats_sorted":[],"title_canon_sha256":"41f26fdd53dc2c1071a7ab1fa2590188d0d1ae6019c91ee7e84844bc029b3978","abstract_canon_sha256":"8f22b227225e05f279b0f80b2960241633b5d4eee9ab450e8671fa3c2e613a5c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:04:53.407262Z","signature_b64":"y30BXpNjTRz2IUg39oREhLIppGrPRv6BygH6tFxET89Hp3FX8mDWS+WkcoCi5PFdpVbufBXSGAjIO4fdFftGAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f7927516cacfa62b5e925a5f648ea66b5c28451891a5d09ff99c13971b5c07d1","last_reissued_at":"2026-07-05T04:04:53.406861Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:04:53.406861Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DARA: Dynamics-Aware Reward Augmentation in Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Donglin Wang, Hongyin Zhang, Jinxin Liu","submitted_at":"2022-03-13T14:30:55Z","abstract_excerpt":"Offline reinforcement learning algorithms promise to be applicable in settings where a fixed dataset is available and no new experience can be acquired. However, such formulation is inevitably offline-data-hungry and, in practice, collecting a large offline dataset for one specific task over one specific environment is also costly and laborious. In this paper, we thus 1) formulate the offline dynamics adaptation by using (source) offline data collected from another dynamics to relax the requirement for the extensive (target) offline data, 2) characterize the dynamics shift problem in which pri"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.06662","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.06662/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.06662","created_at":"2026-07-05T04:04:53.406917+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.06662v1","created_at":"2026-07-05T04:04:53.406917+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.06662","created_at":"2026-07-05T04:04:53.406917+00:00"},{"alias_kind":"pith_short_12","alias_value":"66JHKFWKZ6TC","created_at":"2026-07-05T04:04:53.406917+00:00"},{"alias_kind":"pith_short_16","alias_value":"66JHKFWKZ6TCWXUS","created_at":"2026-07-05T04:04:53.406917+00:00"},{"alias_kind":"pith_short_8","alias_value":"66JHKFWK","created_at":"2026-07-05T04:04:53.406917+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24810","citing_title":"Cross-Domain Energy-Guided Diffusion Generation for Off-Dynamics Reinforcement Learning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22376","citing_title":"Target-Aligned Bellman Backup for Cross-domain Offline Reinforcement Learning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01862","citing_title":"QHyer: Q-conditioned Hybrid Attention-mamba Transformer for Offline Goal-conditioned RL","ref_index":104,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/66JHKFWKZ6TCWXUSLJPWJDVGNN","json":"https://pith.science/pith/66JHKFWKZ6TCWXUSLJPWJDVGNN.json","graph_json":"https://pith.science/api/pith-number/66JHKFWKZ6TCWXUSLJPWJDVGNN/graph.json","events_json":"https://pith.science/api/pith-number/66JHKFWKZ6TCWXUSLJPWJDVGNN/events.json","paper":"https://pith.science/paper/66JHKFWK"},"agent_actions":{"view_html":"https://pith.science/pith/66JHKFWKZ6TCWXUSLJPWJDVGNN","download_json":"https://pith.science/pith/66JHKFWKZ6TCWXUSLJPWJDVGNN.json","view_paper":"https://pith.science/paper/66JHKFWK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.06662&json=true","fetch_graph":"https://pith.science/api/pith-number/66JHKFWKZ6TCWXUSLJPWJDVGNN/graph.json","fetch_events":"https://pith.science/api/pith-number/66JHKFWKZ6TCWXUSLJPWJDVGNN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/66JHKFWKZ6TCWXUSLJPWJDVGNN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/66JHKFWKZ6TCWXUSLJPWJDVGNN/action/storage_attestation","attest_author":"https://pith.science/pith/66JHKFWKZ6TCWXUSLJPWJDVGNN/action/author_attestation","sign_citation":"https://pith.science/pith/66JHKFWKZ6TCWXUSLJPWJDVGNN/action/citation_signature","submit_replication":"https://pith.science/pith/66JHKFWKZ6TCWXUSLJPWJDVGNN/action/replication_record"}},"created_at":"2026-07-05T04:04:53.406917+00:00","updated_at":"2026-07-05T04:04:53.406917+00:00"}