{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:SM7CI5HEBPUJI6QRVYV7UTZHS3","short_pith_number":"pith:SM7CI5HE","schema_version":"1.0","canonical_sha256":"933e2474e40be8947a11ae2bfa4f2796da7be83446357522f3c2b950147d3297","source":{"kind":"arxiv","id":"2202.02433","version":2},"attestation_state":"computed","paper":{"title":"Versatile Offline Imitation from Observations and Examples via Regularized State-Occupancy Matching","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Andrew Shen, Dinesh Jayaraman, Osbert Bastani, Yecheng Jason Ma","submitted_at":"2022-02-04T23:25:03Z","abstract_excerpt":"We propose State Matching Offline DIstribution Correction Estimation (SMODICE), a novel and versatile regression-based offline imitation learning (IL) algorithm derived via state-occupancy matching. We show that the SMODICE objective admits a simple optimization procedure through an application of Fenchel duality and an analytic solution in tabular MDPs. Without requiring access to expert actions, SMODICE can be effectively applied to three offline IL settings: (i) imitation from observations (IfO), (ii) IfO with dynamics or morphologically mismatched expert, and (iii) example-based reinforcem"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.02433","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-02-04T23:25:03Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4b5c36c8c1566dc91e39fe1bf271a132b652fa0863c4714f446f8f85a5687869","abstract_canon_sha256":"50d35e2aec7075ff4423a0f41a5c2c62667e49bc8add64d533b1921ba8100c67"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:32:55.735093Z","signature_b64":"ZAc5znj6YiRjxtHV9MqDuKvsSxQELsBgP7trQfua7sqchluScIUW9nKmKrt0LJYz3asNCrPIMG44kGb/4WZwAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"933e2474e40be8947a11ae2bfa4f2796da7be83446357522f3c2b950147d3297","last_reissued_at":"2026-07-05T04:32:55.734566Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:32:55.734566Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Versatile Offline Imitation from Observations and Examples via Regularized State-Occupancy Matching","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Andrew Shen, Dinesh Jayaraman, Osbert Bastani, Yecheng Jason Ma","submitted_at":"2022-02-04T23:25:03Z","abstract_excerpt":"We propose State Matching Offline DIstribution Correction Estimation (SMODICE), a novel and versatile regression-based offline imitation learning (IL) algorithm derived via state-occupancy matching. We show that the SMODICE objective admits a simple optimization procedure through an application of Fenchel duality and an analytic solution in tabular MDPs. Without requiring access to expert actions, SMODICE can be effectively applied to three offline IL settings: (i) imitation from observations (IfO), (ii) IfO with dynamics or morphologically mismatched expert, and (iii) example-based reinforcem"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.02433","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.02433/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.02433","created_at":"2026-07-05T04:32:55.734634+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.02433v2","created_at":"2026-07-05T04:32:55.734634+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.02433","created_at":"2026-07-05T04:32:55.734634+00:00"},{"alias_kind":"pith_short_12","alias_value":"SM7CI5HEBPUJ","created_at":"2026-07-05T04:32:55.734634+00:00"},{"alias_kind":"pith_short_16","alias_value":"SM7CI5HEBPUJI6QR","created_at":"2026-07-05T04:32:55.734634+00:00"},{"alias_kind":"pith_short_8","alias_value":"SM7CI5HE","created_at":"2026-07-05T04:32:55.734634+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2210.00030","citing_title":"VIP: Towards Universal Visual Reward and Representation via Value-Implicit Pre-Training","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SM7CI5HEBPUJI6QRVYV7UTZHS3","json":"https://pith.science/pith/SM7CI5HEBPUJI6QRVYV7UTZHS3.json","graph_json":"https://pith.science/api/pith-number/SM7CI5HEBPUJI6QRVYV7UTZHS3/graph.json","events_json":"https://pith.science/api/pith-number/SM7CI5HEBPUJI6QRVYV7UTZHS3/events.json","paper":"https://pith.science/paper/SM7CI5HE"},"agent_actions":{"view_html":"https://pith.science/pith/SM7CI5HEBPUJI6QRVYV7UTZHS3","download_json":"https://pith.science/pith/SM7CI5HEBPUJI6QRVYV7UTZHS3.json","view_paper":"https://pith.science/paper/SM7CI5HE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.02433&json=true","fetch_graph":"https://pith.science/api/pith-number/SM7CI5HEBPUJI6QRVYV7UTZHS3/graph.json","fetch_events":"https://pith.science/api/pith-number/SM7CI5HEBPUJI6QRVYV7UTZHS3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SM7CI5HEBPUJI6QRVYV7UTZHS3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SM7CI5HEBPUJI6QRVYV7UTZHS3/action/storage_attestation","attest_author":"https://pith.science/pith/SM7CI5HEBPUJI6QRVYV7UTZHS3/action/author_attestation","sign_citation":"https://pith.science/pith/SM7CI5HEBPUJI6QRVYV7UTZHS3/action/citation_signature","submit_replication":"https://pith.science/pith/SM7CI5HEBPUJI6QRVYV7UTZHS3/action/replication_record"}},"created_at":"2026-07-05T04:32:55.734634+00:00","updated_at":"2026-07-05T04:32:55.734634+00:00"}