{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AVZW4IFFPL2TAFLBTYOGNOXUWZ","short_pith_number":"pith:AVZW4IFF","schema_version":"1.0","canonical_sha256":"05736e20a57af53015619e1c66baf4b67ad5b532550e60c4e225ad931a06ef21","source":{"kind":"arxiv","id":"2505.00503","version":3},"attestation_state":"computed","paper":{"title":"Variational OOD State Correction for Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Ke Jiang, Wen Jiang, Xiaoyang Tan","submitted_at":"2025-05-01T13:14:07Z","abstract_excerpt":"The performance of Offline reinforcement learning is significantly impacted by the issue of state distributional shift, and out-of-distribution (OOD) state correction is a popular approach to address this problem. In this paper, we propose a novel method named Density-Aware Safety Perception (DASP) for OOD state correction. Specifically, our method encourages the agent to prioritize actions that lead to outcomes with higher data density, thereby promoting its operation within or the return to in-distribution (safe) regions. To achieve this, we optimize the objective within a variational framew"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.00503","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-01T13:14:07Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"2befb230b6ac32c46c5694c3cf98e9d10e3bfe5b95e25c0811948d1826c7b012","abstract_canon_sha256":"e4a3cff7368a1ea1b20fc4735ee87e23af348b5a3f26f5f1868e8cff43f39390"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:33:23.347093Z","signature_b64":"qsALs7+V3+VQxC9RLPSeVB4eYJ76ZV9L1H3WWAk1MEMn0/bNyUsp/Eu+3Z8MI+ROWF9bMUz2TDwk6NfYGoC2Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"05736e20a57af53015619e1c66baf4b67ad5b532550e60c4e225ad931a06ef21","last_reissued_at":"2026-07-05T11:33:23.346623Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:33:23.346623Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Variational OOD State Correction for Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Ke Jiang, Wen Jiang, Xiaoyang Tan","submitted_at":"2025-05-01T13:14:07Z","abstract_excerpt":"The performance of Offline reinforcement learning is significantly impacted by the issue of state distributional shift, and out-of-distribution (OOD) state correction is a popular approach to address this problem. In this paper, we propose a novel method named Density-Aware Safety Perception (DASP) for OOD state correction. Specifically, our method encourages the agent to prioritize actions that lead to outcomes with higher data density, thereby promoting its operation within or the return to in-distribution (safe) regions. To achieve this, we optimize the objective within a variational framew"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.00503","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.00503/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.00503","created_at":"2026-07-05T11:33:23.346682+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.00503v3","created_at":"2026-07-05T11:33:23.346682+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.00503","created_at":"2026-07-05T11:33:23.346682+00:00"},{"alias_kind":"pith_short_12","alias_value":"AVZW4IFFPL2T","created_at":"2026-07-05T11:33:23.346682+00:00"},{"alias_kind":"pith_short_16","alias_value":"AVZW4IFFPL2TAFLB","created_at":"2026-07-05T11:33:23.346682+00:00"},{"alias_kind":"pith_short_8","alias_value":"AVZW4IFF","created_at":"2026-07-05T11:33:23.346682+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AVZW4IFFPL2TAFLBTYOGNOXUWZ","json":"https://pith.science/pith/AVZW4IFFPL2TAFLBTYOGNOXUWZ.json","graph_json":"https://pith.science/api/pith-number/AVZW4IFFPL2TAFLBTYOGNOXUWZ/graph.json","events_json":"https://pith.science/api/pith-number/AVZW4IFFPL2TAFLBTYOGNOXUWZ/events.json","paper":"https://pith.science/paper/AVZW4IFF"},"agent_actions":{"view_html":"https://pith.science/pith/AVZW4IFFPL2TAFLBTYOGNOXUWZ","download_json":"https://pith.science/pith/AVZW4IFFPL2TAFLBTYOGNOXUWZ.json","view_paper":"https://pith.science/paper/AVZW4IFF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.00503&json=true","fetch_graph":"https://pith.science/api/pith-number/AVZW4IFFPL2TAFLBTYOGNOXUWZ/graph.json","fetch_events":"https://pith.science/api/pith-number/AVZW4IFFPL2TAFLBTYOGNOXUWZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AVZW4IFFPL2TAFLBTYOGNOXUWZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AVZW4IFFPL2TAFLBTYOGNOXUWZ/action/storage_attestation","attest_author":"https://pith.science/pith/AVZW4IFFPL2TAFLBTYOGNOXUWZ/action/author_attestation","sign_citation":"https://pith.science/pith/AVZW4IFFPL2TAFLBTYOGNOXUWZ/action/citation_signature","submit_replication":"https://pith.science/pith/AVZW4IFFPL2TAFLBTYOGNOXUWZ/action/replication_record"}},"created_at":"2026-07-05T11:33:23.346682+00:00","updated_at":"2026-07-05T11:33:23.346682+00:00"}