{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HXQWUTAKQ76OZNKEEUXJLFK53S","short_pith_number":"pith:HXQWUTAK","schema_version":"1.0","canonical_sha256":"3de16a4c0a87fcecb544252e95955ddc940ddafd054417105fe9a30c648f01dd","source":{"kind":"arxiv","id":"2302.08560","version":3},"attestation_state":"computed","paper":{"title":"Dual RL: Unification and New Methods for Reinforcement and Imitation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Amy Zhang, Harshit Sikchi, Qinqing Zheng, Scott Niekum","submitted_at":"2023-02-16T20:10:06Z","abstract_excerpt":"The goal of reinforcement learning (RL) is to find a policy that maximizes the expected cumulative return. It has been shown that this objective can be represented as an optimization problem of state-action visitation distribution under linear constraints. The dual problem of this formulation, which we refer to as dual RL, is unconstrained and easier to optimize. In this work, we first cast several state-of-the-art offline RL and offline imitation learning (IL) algorithms as instances of dual RL approaches with shared structures. Such unification allows us to identify the root cause of the sho"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.08560","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-02-16T20:10:06Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"26c8d89119af6fb79505c10dcf553b8cf48ae2dcc24732ea107746e386f164c6","abstract_canon_sha256":"4da588c2305a27ff7545b2c9884fd2130825eea23f9183306d6e375f14d65bf9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:37:48.243458Z","signature_b64":"7oeq5UiuaWcyXsx3doZAlraTpASUDJXYQ0vxQLXHzYSIbNoQttcbEw9VULjfzna6RGmjfTbVZvVtvxuRnEQkAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3de16a4c0a87fcecb544252e95955ddc940ddafd054417105fe9a30c648f01dd","last_reissued_at":"2026-07-05T07:37:48.243024Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:37:48.243024Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dual RL: Unification and New Methods for Reinforcement and Imitation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Amy Zhang, Harshit Sikchi, Qinqing Zheng, Scott Niekum","submitted_at":"2023-02-16T20:10:06Z","abstract_excerpt":"The goal of reinforcement learning (RL) is to find a policy that maximizes the expected cumulative return. It has been shown that this objective can be represented as an optimization problem of state-action visitation distribution under linear constraints. The dual problem of this formulation, which we refer to as dual RL, is unconstrained and easier to optimize. In this work, we first cast several state-of-the-art offline RL and offline imitation learning (IL) algorithms as instances of dual RL approaches with shared structures. Such unification allows us to identify the root cause of the sho"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.08560","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.08560/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.08560","created_at":"2026-07-05T07:37:48.243080+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.08560v3","created_at":"2026-07-05T07:37:48.243080+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.08560","created_at":"2026-07-05T07:37:48.243080+00:00"},{"alias_kind":"pith_short_12","alias_value":"HXQWUTAKQ76O","created_at":"2026-07-05T07:37:48.243080+00:00"},{"alias_kind":"pith_short_16","alias_value":"HXQWUTAKQ76OZNKE","created_at":"2026-07-05T07:37:48.243080+00:00"},{"alias_kind":"pith_short_8","alias_value":"HXQWUTAK","created_at":"2026-07-05T07:37:48.243080+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2408.08812","citing_title":"TRAM: Test-Time Risk Adaptation with Mixture of Agents","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01663","citing_title":"Towards Efficient and Expressive Offline RL via Flow-Anchored Noise-conditioned Q-Learning","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01862","citing_title":"QHyer: Q-conditioned Hybrid Attention-mamba Transformer for Offline Goal-conditioned RL","ref_index":185,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17919","citing_title":"Fisher Decorator: Refining Flow Policy via a Local Transport Map","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20627","citing_title":"Occupancy Reward Shaping: Improving Credit Assignment for Offline Goal-Conditioned Reinforcement Learning","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HXQWUTAKQ76OZNKEEUXJLFK53S","json":"https://pith.science/pith/HXQWUTAKQ76OZNKEEUXJLFK53S.json","graph_json":"https://pith.science/api/pith-number/HXQWUTAKQ76OZNKEEUXJLFK53S/graph.json","events_json":"https://pith.science/api/pith-number/HXQWUTAKQ76OZNKEEUXJLFK53S/events.json","paper":"https://pith.science/paper/HXQWUTAK"},"agent_actions":{"view_html":"https://pith.science/pith/HXQWUTAKQ76OZNKEEUXJLFK53S","download_json":"https://pith.science/pith/HXQWUTAKQ76OZNKEEUXJLFK53S.json","view_paper":"https://pith.science/paper/HXQWUTAK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.08560&json=true","fetch_graph":"https://pith.science/api/pith-number/HXQWUTAKQ76OZNKEEUXJLFK53S/graph.json","fetch_events":"https://pith.science/api/pith-number/HXQWUTAKQ76OZNKEEUXJLFK53S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HXQWUTAKQ76OZNKEEUXJLFK53S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HXQWUTAKQ76OZNKEEUXJLFK53S/action/storage_attestation","attest_author":"https://pith.science/pith/HXQWUTAKQ76OZNKEEUXJLFK53S/action/author_attestation","sign_citation":"https://pith.science/pith/HXQWUTAKQ76OZNKEEUXJLFK53S/action/citation_signature","submit_replication":"https://pith.science/pith/HXQWUTAKQ76OZNKEEUXJLFK53S/action/replication_record"}},"created_at":"2026-07-05T07:37:48.243080+00:00","updated_at":"2026-07-05T07:37:48.243080+00:00"}