{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MC6Q3DHRTR34ATCH7KOTZ4RKS6","short_pith_number":"pith:MC6Q3DHR","schema_version":"1.0","canonical_sha256":"60bd0d8cf19c77c04c47fa9d3cf22a978cac99806d788ac513d7379dbc9a0ab7","source":{"kind":"arxiv","id":"2409.20521","version":1},"attestation_state":"computed","paper":{"title":"Upper and Lower Bounds for Distributionally Robust Off-Dynamics Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Pan Xu, Weixin Wang, Zhishuai Liu","submitted_at":"2024-09-30T17:21:15Z","abstract_excerpt":"We study off-dynamics Reinforcement Learning (RL), where the policy training and deployment environments are different. To deal with this environmental perturbation, we focus on learning policies robust to uncertainties in transition dynamics under the framework of distributionally robust Markov decision processes (DRMDPs), where the nominal and perturbed dynamics are linear Markov Decision Processes. We propose a novel algorithm We-DRIVE-U that enjoys an average suboptimality $\\widetilde{\\mathcal{O}}\\big({d H \\cdot \\min \\{1/{\\rho}, H\\}/\\sqrt{K} }\\big)$, where $K$ is the number of episodes, $H"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.20521","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-30T17:21:15Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"fb3cf7e0166b20a45caab0952f770a8c0a9fe364bdbce916ccd6bdc03a512afd","abstract_canon_sha256":"efbd1b1e8a7891ef8a9470026db1a88969f3d6bc072f1ba747ab639de301bb0b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:13:44.779854Z","signature_b64":"sv9haaovn7Aw3aD+FFXEKcfcGPMply9XBImBG9KH0L9DjSHTUfWlK/iyBveT2GpERZs3DEfHcUiJbBzcbVoOAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"60bd0d8cf19c77c04c47fa9d3cf22a978cac99806d788ac513d7379dbc9a0ab7","last_reissued_at":"2026-07-05T09:13:44.779443Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:13:44.779443Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Upper and Lower Bounds for Distributionally Robust Off-Dynamics Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Pan Xu, Weixin Wang, Zhishuai Liu","submitted_at":"2024-09-30T17:21:15Z","abstract_excerpt":"We study off-dynamics Reinforcement Learning (RL), where the policy training and deployment environments are different. To deal with this environmental perturbation, we focus on learning policies robust to uncertainties in transition dynamics under the framework of distributionally robust Markov decision processes (DRMDPs), where the nominal and perturbed dynamics are linear Markov Decision Processes. We propose a novel algorithm We-DRIVE-U that enjoys an average suboptimality $\\widetilde{\\mathcal{O}}\\big({d H \\cdot \\min \\{1/{\\rho}, H\\}/\\sqrt{K} }\\big)$, where $K$ is the number of episodes, $H"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.20521","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.20521/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.20521","created_at":"2026-07-05T09:13:44.779504+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.20521v1","created_at":"2026-07-05T09:13:44.779504+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.20521","created_at":"2026-07-05T09:13:44.779504+00:00"},{"alias_kind":"pith_short_12","alias_value":"MC6Q3DHRTR34","created_at":"2026-07-05T09:13:44.779504+00:00"},{"alias_kind":"pith_short_16","alias_value":"MC6Q3DHRTR34ATCH","created_at":"2026-07-05T09:13:44.779504+00:00"},{"alias_kind":"pith_short_8","alias_value":"MC6Q3DHR","created_at":"2026-07-05T09:13:44.779504+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24810","citing_title":"Cross-Domain Energy-Guided Diffusion Generation for Off-Dynamics Reinforcement Learning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26627","citing_title":"Breaking the Epistemic Trap: Active Perception Under Compound Uncertainty","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26627","citing_title":"Breaking the Epistemic Trap: Active Perception Under Compound Uncertainty","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MC6Q3DHRTR34ATCH7KOTZ4RKS6","json":"https://pith.science/pith/MC6Q3DHRTR34ATCH7KOTZ4RKS6.json","graph_json":"https://pith.science/api/pith-number/MC6Q3DHRTR34ATCH7KOTZ4RKS6/graph.json","events_json":"https://pith.science/api/pith-number/MC6Q3DHRTR34ATCH7KOTZ4RKS6/events.json","paper":"https://pith.science/paper/MC6Q3DHR"},"agent_actions":{"view_html":"https://pith.science/pith/MC6Q3DHRTR34ATCH7KOTZ4RKS6","download_json":"https://pith.science/pith/MC6Q3DHRTR34ATCH7KOTZ4RKS6.json","view_paper":"https://pith.science/paper/MC6Q3DHR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.20521&json=true","fetch_graph":"https://pith.science/api/pith-number/MC6Q3DHRTR34ATCH7KOTZ4RKS6/graph.json","fetch_events":"https://pith.science/api/pith-number/MC6Q3DHRTR34ATCH7KOTZ4RKS6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MC6Q3DHRTR34ATCH7KOTZ4RKS6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MC6Q3DHRTR34ATCH7KOTZ4RKS6/action/storage_attestation","attest_author":"https://pith.science/pith/MC6Q3DHRTR34ATCH7KOTZ4RKS6/action/author_attestation","sign_citation":"https://pith.science/pith/MC6Q3DHRTR34ATCH7KOTZ4RKS6/action/citation_signature","submit_replication":"https://pith.science/pith/MC6Q3DHRTR34ATCH7KOTZ4RKS6/action/replication_record"}},"created_at":"2026-07-05T09:13:44.779504+00:00","updated_at":"2026-07-05T09:13:44.779504+00:00"}