{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:HXQWUTAKQ76OZNKEEUXJLFK53S","short_pith_number":"pith:HXQWUTAK","canonical_record":{"source":{"id":"2302.08560","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-02-16T20:10:06Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"26c8d89119af6fb79505c10dcf553b8cf48ae2dcc24732ea107746e386f164c6","abstract_canon_sha256":"4da588c2305a27ff7545b2c9884fd2130825eea23f9183306d6e375f14d65bf9"},"schema_version":"1.0"},"canonical_sha256":"3de16a4c0a87fcecb544252e95955ddc940ddafd054417105fe9a30c648f01dd","source":{"kind":"arxiv","id":"2302.08560","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2302.08560","created_at":"2026-07-05T07:37:48Z"},{"alias_kind":"arxiv_version","alias_value":"2302.08560v3","created_at":"2026-07-05T07:37:48Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.08560","created_at":"2026-07-05T07:37:48Z"},{"alias_kind":"pith_short_12","alias_value":"HXQWUTAKQ76O","created_at":"2026-07-05T07:37:48Z"},{"alias_kind":"pith_short_16","alias_value":"HXQWUTAKQ76OZNKE","created_at":"2026-07-05T07:37:48Z"},{"alias_kind":"pith_short_8","alias_value":"HXQWUTAK","created_at":"2026-07-05T07:37:48Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:HXQWUTAKQ76OZNKEEUXJLFK53S","target":"record","payload":{"canonical_record":{"source":{"id":"2302.08560","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-02-16T20:10:06Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"26c8d89119af6fb79505c10dcf553b8cf48ae2dcc24732ea107746e386f164c6","abstract_canon_sha256":"4da588c2305a27ff7545b2c9884fd2130825eea23f9183306d6e375f14d65bf9"},"schema_version":"1.0"},"canonical_sha256":"3de16a4c0a87fcecb544252e95955ddc940ddafd054417105fe9a30c648f01dd","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:37:48.243458Z","signature_b64":"7oeq5UiuaWcyXsx3doZAlraTpASUDJXYQ0vxQLXHzYSIbNoQttcbEw9VULjfzna6RGmjfTbVZvVtvxuRnEQkAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3de16a4c0a87fcecb544252e95955ddc940ddafd054417105fe9a30c648f01dd","last_reissued_at":"2026-07-05T07:37:48.243024Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:37:48.243024Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2302.08560","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:37:48Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4XjilpaA4lGlnmtwlTGWXiZ3fRMO/5TpaCAKhrL+TeIjF474vW5S8V6hSEKBPpGVjlKTiIDS1Ywy2TCs59FYBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T19:14:16.949348Z"},"content_sha256":"25eb20a4d7eb10171e94898535efaa06339d909693e41acbf18cd7cd7467852d","schema_version":"1.0","event_id":"sha256:25eb20a4d7eb10171e94898535efaa06339d909693e41acbf18cd7cd7467852d"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:HXQWUTAKQ76OZNKEEUXJLFK53S","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Dual RL: Unification and New Methods for Reinforcement and Imitation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Amy Zhang, Harshit Sikchi, Qinqing Zheng, Scott Niekum","submitted_at":"2023-02-16T20:10:06Z","abstract_excerpt":"The goal of reinforcement learning (RL) is to find a policy that maximizes the expected cumulative return. It has been shown that this objective can be represented as an optimization problem of state-action visitation distribution under linear constraints. The dual problem of this formulation, which we refer to as dual RL, is unconstrained and easier to optimize. In this work, we first cast several state-of-the-art offline RL and offline imitation learning (IL) algorithms as instances of dual RL approaches with shared structures. Such unification allows us to identify the root cause of the sho"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.08560","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.08560/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:37:48Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"bomRkoblSjW6tL0dr3rn48S9aVXuc/GG+L6DkRJeiTze3tN65m8MA8vCfxd2D7xbnxhyE/n1IeroHm3uE/0MDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T19:14:16.950224Z"},"content_sha256":"1924c1ffe624da86959533d85b2f3a4a1d1c7760b125ab823383eaba5b94c5fb","schema_version":"1.0","event_id":"sha256:1924c1ffe624da86959533d85b2f3a4a1d1c7760b125ab823383eaba5b94c5fb"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/HXQWUTAKQ76OZNKEEUXJLFK53S/bundle.json","state_url":"https://pith.science/pith/HXQWUTAKQ76OZNKEEUXJLFK53S/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/HXQWUTAKQ76OZNKEEUXJLFK53S/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T19:14:16Z","links":{"resolver":"https://pith.science/pith/HXQWUTAKQ76OZNKEEUXJLFK53S","bundle":"https://pith.science/pith/HXQWUTAKQ76OZNKEEUXJLFK53S/bundle.json","state":"https://pith.science/pith/HXQWUTAKQ76OZNKEEUXJLFK53S/state.json","well_known_bundle":"https://pith.science/.well-known/pith/HXQWUTAKQ76OZNKEEUXJLFK53S/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:HXQWUTAKQ76OZNKEEUXJLFK53S","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"4da588c2305a27ff7545b2c9884fd2130825eea23f9183306d6e375f14d65bf9","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-02-16T20:10:06Z","title_canon_sha256":"26c8d89119af6fb79505c10dcf553b8cf48ae2dcc24732ea107746e386f164c6"},"schema_version":"1.0","source":{"id":"2302.08560","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2302.08560","created_at":"2026-07-05T07:37:48Z"},{"alias_kind":"arxiv_version","alias_value":"2302.08560v3","created_at":"2026-07-05T07:37:48Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.08560","created_at":"2026-07-05T07:37:48Z"},{"alias_kind":"pith_short_12","alias_value":"HXQWUTAKQ76O","created_at":"2026-07-05T07:37:48Z"},{"alias_kind":"pith_short_16","alias_value":"HXQWUTAKQ76OZNKE","created_at":"2026-07-05T07:37:48Z"},{"alias_kind":"pith_short_8","alias_value":"HXQWUTAK","created_at":"2026-07-05T07:37:48Z"}],"graph_snapshots":[{"event_id":"sha256:1924c1ffe624da86959533d85b2f3a4a1d1c7760b125ab823383eaba5b94c5fb","target":"graph","created_at":"2026-07-05T07:37:48Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2302.08560/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The goal of reinforcement learning (RL) is to find a policy that maximizes the expected cumulative return. It has been shown that this objective can be represented as an optimization problem of state-action visitation distribution under linear constraints. The dual problem of this formulation, which we refer to as dual RL, is unconstrained and easier to optimize. In this work, we first cast several state-of-the-art offline RL and offline imitation learning (IL) algorithms as instances of dual RL approaches with shared structures. Such unification allows us to identify the root cause of the sho","authors_text":"Amy Zhang, Harshit Sikchi, Qinqing Zheng, Scott Niekum","cross_cats":["cs.AI","cs.RO"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-02-16T20:10:06Z","title":"Dual RL: Unification and New Methods for Reinforcement and Imitation Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.08560","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:25eb20a4d7eb10171e94898535efaa06339d909693e41acbf18cd7cd7467852d","target":"record","created_at":"2026-07-05T07:37:48Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"4da588c2305a27ff7545b2c9884fd2130825eea23f9183306d6e375f14d65bf9","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-02-16T20:10:06Z","title_canon_sha256":"26c8d89119af6fb79505c10dcf553b8cf48ae2dcc24732ea107746e386f164c6"},"schema_version":"1.0","source":{"id":"2302.08560","kind":"arxiv","version":3}},"canonical_sha256":"3de16a4c0a87fcecb544252e95955ddc940ddafd054417105fe9a30c648f01dd","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3de16a4c0a87fcecb544252e95955ddc940ddafd054417105fe9a30c648f01dd","first_computed_at":"2026-07-05T07:37:48.243024Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:37:48.243024Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"7oeq5UiuaWcyXsx3doZAlraTpASUDJXYQ0vxQLXHzYSIbNoQttcbEw9VULjfzna6RGmjfTbVZvVtvxuRnEQkAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T07:37:48.243458Z","signed_message":"canonical_sha256_bytes"},"source_id":"2302.08560","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:25eb20a4d7eb10171e94898535efaa06339d909693e41acbf18cd7cd7467852d","sha256:1924c1ffe624da86959533d85b2f3a4a1d1c7760b125ab823383eaba5b94c5fb"],"state_sha256":"935c658f7f38d4d3f14702fd2431d0ac1a175e7d3c05101d5b886999b2e60ed3"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"V2agwsMijwqZy1wGRkQO7bjTlKQ8Np1fAjEuSnCv98eopQfgFLYV3wa+YVz5NYSOmwUEnLICr4OZwcQZmGR8Cw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T19:14:16.975542Z","bundle_sha256":"1febc261f0ab39982b8399e626492a23cd3a805536d4d8c848d06a4f553694b1"}}