{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:4HTOCR6GMUHNLHHKTHW2KPTZZL","short_pith_number":"pith:4HTOCR6G","schema_version":"1.0","canonical_sha256":"e1e6e147c6650ed59cea99eda53e79caed6f4dd736aac75e63287e5439ae1b7f","source":{"kind":"arxiv","id":"2106.10783","version":1},"attestation_state":"computed","paper":{"title":"OptiDICE: Offline Policy Optimization via Stationary Distribution Correction Estimation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Byung-Jun Lee, Joelle Pineau, Jongmin Lee, Kee-Eung Kim, Wonseok Jeon","submitted_at":"2021-06-21T00:43:30Z","abstract_excerpt":"We consider the offline reinforcement learning (RL) setting where the agent aims to optimize the policy solely from the data without further environment interactions. In offline RL, the distributional shift becomes the primary source of difficulty, which arises from the deviation of the target policy being optimized from the behavior policy used for data collection. This typically causes overestimation of action values, which poses severe problems for model-free algorithms that use bootstrapping. To mitigate the problem, prior offline RL algorithms often used sophisticated techniques that enco"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.10783","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2021-06-21T00:43:30Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2cf6caad23ffb3d00c3e587b9641b49e076a15befb405d689e91b043da145d3e","abstract_canon_sha256":"356ced242c047213cbf74fbd795bc2745a19eed65a5f94dd013d566dc04be498"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:50:53.307703Z","signature_b64":"RYz7Y73lQy6Y1JXxjv6stazuPj4PUpRFC5VEYRvy+V+rkXkwk4VAFqnOq0bXRDKemDB3zeZ6B1bKScRpBr4tBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e1e6e147c6650ed59cea99eda53e79caed6f4dd736aac75e63287e5439ae1b7f","last_reissued_at":"2026-07-05T02:50:53.307257Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:50:53.307257Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OptiDICE: Offline Policy Optimization via Stationary Distribution Correction Estimation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Byung-Jun Lee, Joelle Pineau, Jongmin Lee, Kee-Eung Kim, Wonseok Jeon","submitted_at":"2021-06-21T00:43:30Z","abstract_excerpt":"We consider the offline reinforcement learning (RL) setting where the agent aims to optimize the policy solely from the data without further environment interactions. In offline RL, the distributional shift becomes the primary source of difficulty, which arises from the deviation of the target policy being optimized from the behavior policy used for data collection. This typically causes overestimation of action values, which poses severe problems for model-free algorithms that use bootstrapping. To mitigate the problem, prior offline RL algorithms often used sophisticated techniques that enco"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.10783","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.10783/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.10783","created_at":"2026-07-05T02:50:53.307324+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.10783v1","created_at":"2026-07-05T02:50:53.307324+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.10783","created_at":"2026-07-05T02:50:53.307324+00:00"},{"alias_kind":"pith_short_12","alias_value":"4HTOCR6GMUHN","created_at":"2026-07-05T02:50:53.307324+00:00"},{"alias_kind":"pith_short_16","alias_value":"4HTOCR6GMUHNLHHK","created_at":"2026-07-05T02:50:53.307324+00:00"},{"alias_kind":"pith_short_8","alias_value":"4HTOCR6G","created_at":"2026-07-05T02:50:53.307324+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4HTOCR6GMUHNLHHKTHW2KPTZZL","json":"https://pith.science/pith/4HTOCR6GMUHNLHHKTHW2KPTZZL.json","graph_json":"https://pith.science/api/pith-number/4HTOCR6GMUHNLHHKTHW2KPTZZL/graph.json","events_json":"https://pith.science/api/pith-number/4HTOCR6GMUHNLHHKTHW2KPTZZL/events.json","paper":"https://pith.science/paper/4HTOCR6G"},"agent_actions":{"view_html":"https://pith.science/pith/4HTOCR6GMUHNLHHKTHW2KPTZZL","download_json":"https://pith.science/pith/4HTOCR6GMUHNLHHKTHW2KPTZZL.json","view_paper":"https://pith.science/paper/4HTOCR6G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.10783&json=true","fetch_graph":"https://pith.science/api/pith-number/4HTOCR6GMUHNLHHKTHW2KPTZZL/graph.json","fetch_events":"https://pith.science/api/pith-number/4HTOCR6GMUHNLHHKTHW2KPTZZL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4HTOCR6GMUHNLHHKTHW2KPTZZL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4HTOCR6GMUHNLHHKTHW2KPTZZL/action/storage_attestation","attest_author":"https://pith.science/pith/4HTOCR6GMUHNLHHKTHW2KPTZZL/action/author_attestation","sign_citation":"https://pith.science/pith/4HTOCR6GMUHNLHHKTHW2KPTZZL/action/citation_signature","submit_replication":"https://pith.science/pith/4HTOCR6GMUHNLHHKTHW2KPTZZL/action/replication_record"}},"created_at":"2026-07-05T02:50:53.307324+00:00","updated_at":"2026-07-05T02:50:53.307324+00:00"}