{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:R3Y54WME75QGERPQZY5YLLBJJV","short_pith_number":"pith:R3Y54WME","schema_version":"1.0","canonical_sha256":"8ef1de5984ff606245f0ce3b85ac294d4c6225860dd6b7ae74827e31afb84776","source":{"kind":"arxiv","id":"2401.10700","version":1},"attestation_state":"computed","paper":{"title":"Safe Offline Reinforcement Learning with Feasibility-Guided Diffusion Model","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Dongjie Yu, Jianxiong Li, Jingjing Liu, Shengbo Eben Li, Xianyuan Zhan, Yinan Zheng, Yujie Yang","submitted_at":"2024-01-19T14:05:09Z","abstract_excerpt":"Safe offline RL is a promising way to bypass risky online interactions towards safe policy learning. Most existing methods only enforce soft constraints, i.e., constraining safety violations in expectation below thresholds predetermined. This can lead to potentially unsafe outcomes, thus unacceptable in safety-critical scenarios. An alternative is to enforce the hard constraint of zero violation. However, this can be challenging in offline setting, as it needs to strike the right balance among three highly intricate and correlated aspects: safety constraint satisfaction, reward maximization, a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.10700","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-19T14:05:09Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"72a5f63be89628cf65d67977b0be7dfcd68a9044643cf5817ad79e8ef8f2c476","abstract_canon_sha256":"eaf08d1e125662d9d4662d18b5406283b92a9e7cb4bb4f6c48c787f848661f85"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:35:30.245478Z","signature_b64":"2p/HNvi3ksUot12qhhz5fHbtJZ2kDpX7GfbnDBBHkIqVnapj9Oc4OUX2adPMBLTEAmHl5UfTytK0jnLL7DGKDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8ef1de5984ff606245f0ce3b85ac294d4c6225860dd6b7ae74827e31afb84776","last_reissued_at":"2026-07-05T07:35:30.244960Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:35:30.244960Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safe Offline Reinforcement Learning with Feasibility-Guided Diffusion Model","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Dongjie Yu, Jianxiong Li, Jingjing Liu, Shengbo Eben Li, Xianyuan Zhan, Yinan Zheng, Yujie Yang","submitted_at":"2024-01-19T14:05:09Z","abstract_excerpt":"Safe offline RL is a promising way to bypass risky online interactions towards safe policy learning. Most existing methods only enforce soft constraints, i.e., constraining safety violations in expectation below thresholds predetermined. This can lead to potentially unsafe outcomes, thus unacceptable in safety-critical scenarios. An alternative is to enforce the hard constraint of zero violation. However, this can be challenging in offline setting, as it needs to strike the right balance among three highly intricate and correlated aspects: safety constraint satisfaction, reward maximization, a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.10700","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.10700/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.10700","created_at":"2026-07-05T07:35:30.245027+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.10700v1","created_at":"2026-07-05T07:35:30.245027+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.10700","created_at":"2026-07-05T07:35:30.245027+00:00"},{"alias_kind":"pith_short_12","alias_value":"R3Y54WME75QG","created_at":"2026-07-05T07:35:30.245027+00:00"},{"alias_kind":"pith_short_16","alias_value":"R3Y54WME75QGERPQ","created_at":"2026-07-05T07:35:30.245027+00:00"},{"alias_kind":"pith_short_8","alias_value":"R3Y54WME","created_at":"2026-07-05T07:35:30.245027+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08312","citing_title":"Neuro-Symbolic Injection of LTLf Constraints in Autoregressive Reinforcement Learning Policies","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02924","citing_title":"How Does the Lagrangian Guide Safe Reinforcement Learning through Diffusion Models?","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02777","citing_title":"Decoupled Guidance Diffusion for Adaptive Offline Safe Reinforcement Learning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09035","citing_title":"Advantage-Guided Diffusion for Model-Based Reinforcement Learning","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R3Y54WME75QGERPQZY5YLLBJJV","json":"https://pith.science/pith/R3Y54WME75QGERPQZY5YLLBJJV.json","graph_json":"https://pith.science/api/pith-number/R3Y54WME75QGERPQZY5YLLBJJV/graph.json","events_json":"https://pith.science/api/pith-number/R3Y54WME75QGERPQZY5YLLBJJV/events.json","paper":"https://pith.science/paper/R3Y54WME"},"agent_actions":{"view_html":"https://pith.science/pith/R3Y54WME75QGERPQZY5YLLBJJV","download_json":"https://pith.science/pith/R3Y54WME75QGERPQZY5YLLBJJV.json","view_paper":"https://pith.science/paper/R3Y54WME","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.10700&json=true","fetch_graph":"https://pith.science/api/pith-number/R3Y54WME75QGERPQZY5YLLBJJV/graph.json","fetch_events":"https://pith.science/api/pith-number/R3Y54WME75QGERPQZY5YLLBJJV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R3Y54WME75QGERPQZY5YLLBJJV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R3Y54WME75QGERPQZY5YLLBJJV/action/storage_attestation","attest_author":"https://pith.science/pith/R3Y54WME75QGERPQZY5YLLBJJV/action/author_attestation","sign_citation":"https://pith.science/pith/R3Y54WME75QGERPQZY5YLLBJJV/action/citation_signature","submit_replication":"https://pith.science/pith/R3Y54WME75QGERPQZY5YLLBJJV/action/replication_record"}},"created_at":"2026-07-05T07:35:30.245027+00:00","updated_at":"2026-07-05T07:35:30.245027+00:00"}