{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:UMOFOXPDVSOXAN4KQWDWWF4LVO","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"532e2e45fddbde37865a3d6b5dbf7a47443268534d99771eb4f2923db8d39f74","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-29T03:19:59Z","title_canon_sha256":"db8730fe6a4b59a30bc45205f2ccfb8104e57c064eb8f632b209b7fd414def1e"},"schema_version":"1.0","source":{"id":"2405.18729","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.18729","created_at":"2026-07-05T08:24:29Z"},{"alias_kind":"arxiv_version","alias_value":"2405.18729v1","created_at":"2026-07-05T08:24:29Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.18729","created_at":"2026-07-05T08:24:29Z"},{"alias_kind":"pith_short_12","alias_value":"UMOFOXPDVSOX","created_at":"2026-07-05T08:24:29Z"},{"alias_kind":"pith_short_16","alias_value":"UMOFOXPDVSOXAN4K","created_at":"2026-07-05T08:24:29Z"},{"alias_kind":"pith_short_8","alias_value":"UMOFOXPD","created_at":"2026-07-05T08:24:29Z"}],"graph_snapshots":[{"event_id":"sha256:e9f3f70a5ebd4a8eb3ea3cb18d95f57b5843d946beffaae47fea0e8f1895456c","target":"graph","created_at":"2026-07-05T08:24:29Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2405.18729/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Offline reinforcement learning (RL) aims to learn optimal policies from previously collected datasets. Recently, due to their powerful representational capabilities, diffusion models have shown significant potential as policy models for offline RL issues. However, previous offline RL algorithms based on diffusion policies generally adopt weighted regression to improve the policy. This approach optimizes the policy only using the collected actions and is sensitive to Q-values, which limits the potential for further performance enhancement. To this end, we propose a novel preferred-action-optimi","authors_text":"Dongjiang Li, Jiayi Guan, Lei Sun, Lin Zhao, Lusong Li, Tianle Zhang, Xiaodong He, Xuelong Wei, Yihang Li, Yue Chen, Zecui Zeng","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-29T03:19:59Z","title":"Preferred-Action-Optimized Diffusion Policies for Offline Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.18729","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:18359c3b6d4fc71128e90f885f936e376aae56e39147aa6144a85473c064ff59","target":"record","created_at":"2026-07-05T08:24:29Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"532e2e45fddbde37865a3d6b5dbf7a47443268534d99771eb4f2923db8d39f74","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-29T03:19:59Z","title_canon_sha256":"db8730fe6a4b59a30bc45205f2ccfb8104e57c064eb8f632b209b7fd414def1e"},"schema_version":"1.0","source":{"id":"2405.18729","kind":"arxiv","version":1}},"canonical_sha256":"a31c575de3ac9d70378a85876b178baba5dab98a43fdee7c6682d7ec715af293","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"a31c575de3ac9d70378a85876b178baba5dab98a43fdee7c6682d7ec715af293","first_computed_at":"2026-07-05T08:24:29.455622Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:24:29.455622Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"R+Tw6n32sVyEcmbVeWHlzPqvhcMs0nF8vLwRj8zkIacFapwqcMO9t/dIvzh07bPxGmb8xvlu+UZT6Kpq1xZGBg==","signature_status":"signed_v1","signed_at":"2026-07-05T08:24:29.456158Z","signed_message":"canonical_sha256_bytes"},"source_id":"2405.18729","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:18359c3b6d4fc71128e90f885f936e376aae56e39147aa6144a85473c064ff59","sha256:e9f3f70a5ebd4a8eb3ea3cb18d95f57b5843d946beffaae47fea0e8f1895456c"],"state_sha256":"3c3e03e265d5bd42022e75dca460ec26a9a746020ecd9facfb439d9a1ab57020"}