{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LPWQEPXTNSKOFYG3BRZT7EV65U","short_pith_number":"pith:LPWQEPXT","schema_version":"1.0","canonical_sha256":"5bed023ef36c94e2e0db0c733f92beed0d3a709e4e1d9f2f683b87bf4aedc909","source":{"kind":"arxiv","id":"2311.01331","version":3},"attestation_state":"computed","paper":{"title":"Offline Imitation from Observation via Primal Wasserstein State Occupancy Matching","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alexander G. Schwing, Kai Yan, Yu-Xiong Wang","submitted_at":"2023-11-02T15:41:57Z","abstract_excerpt":"In real-world scenarios, arbitrary interactions with the environment can often be costly, and actions of expert demonstrations are not always available. To reduce the need for both, offline Learning from Observations (LfO) is extensively studied: the agent learns to solve a task given only expert states and task-agnostic non-expert state-action pairs. The state-of-the-art DIstribution Correction Estimation (DICE) methods, as exemplified by SMODICE, minimize the state occupancy divergence between the learner's and empirical expert policies. However, such methods are limited to either $f$-diverg"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.01331","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2023-11-02T15:41:57Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"49b0288787997d93ba4dbc94216263f746283bf1e6454f9b3f17cd7fce26efd5","abstract_canon_sha256":"d59c2ca199621b2c5303ae4bca1f953932d4a3f0aff29cd5580c1555ad9066c8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:29:18.491958Z","signature_b64":"XpBXfeC8m7n0+W81EeqTV+JK/lMlRbiK7c9dsBUaF8rNGxJp2AoYQu0YD4YEKXy4v+OB7Ujsnf8FLiwq0KdiDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5bed023ef36c94e2e0db0c733f92beed0d3a709e4e1d9f2f683b87bf4aedc909","last_reissued_at":"2026-07-05T08:29:18.491552Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:29:18.491552Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Offline Imitation from Observation via Primal Wasserstein State Occupancy Matching","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alexander G. Schwing, Kai Yan, Yu-Xiong Wang","submitted_at":"2023-11-02T15:41:57Z","abstract_excerpt":"In real-world scenarios, arbitrary interactions with the environment can often be costly, and actions of expert demonstrations are not always available. To reduce the need for both, offline Learning from Observations (LfO) is extensively studied: the agent learns to solve a task given only expert states and task-agnostic non-expert state-action pairs. The state-of-the-art DIstribution Correction Estimation (DICE) methods, as exemplified by SMODICE, minimize the state occupancy divergence between the learner's and empirical expert policies. However, such methods are limited to either $f$-diverg"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.01331","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.01331/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.01331","created_at":"2026-07-05T08:29:18.491616+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.01331v3","created_at":"2026-07-05T08:29:18.491616+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.01331","created_at":"2026-07-05T08:29:18.491616+00:00"},{"alias_kind":"pith_short_12","alias_value":"LPWQEPXTNSKO","created_at":"2026-07-05T08:29:18.491616+00:00"},{"alias_kind":"pith_short_16","alias_value":"LPWQEPXTNSKOFYG3","created_at":"2026-07-05T08:29:18.491616+00:00"},{"alias_kind":"pith_short_8","alias_value":"LPWQEPXT","created_at":"2026-07-05T08:29:18.491616+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.02867","citing_title":"Domain-Invariant Per-Frame Feature Extraction for Cross-Domain Imitation Learning with Visual Observations","ref_index":2015,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LPWQEPXTNSKOFYG3BRZT7EV65U","json":"https://pith.science/pith/LPWQEPXTNSKOFYG3BRZT7EV65U.json","graph_json":"https://pith.science/api/pith-number/LPWQEPXTNSKOFYG3BRZT7EV65U/graph.json","events_json":"https://pith.science/api/pith-number/LPWQEPXTNSKOFYG3BRZT7EV65U/events.json","paper":"https://pith.science/paper/LPWQEPXT"},"agent_actions":{"view_html":"https://pith.science/pith/LPWQEPXTNSKOFYG3BRZT7EV65U","download_json":"https://pith.science/pith/LPWQEPXTNSKOFYG3BRZT7EV65U.json","view_paper":"https://pith.science/paper/LPWQEPXT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.01331&json=true","fetch_graph":"https://pith.science/api/pith-number/LPWQEPXTNSKOFYG3BRZT7EV65U/graph.json","fetch_events":"https://pith.science/api/pith-number/LPWQEPXTNSKOFYG3BRZT7EV65U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LPWQEPXTNSKOFYG3BRZT7EV65U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LPWQEPXTNSKOFYG3BRZT7EV65U/action/storage_attestation","attest_author":"https://pith.science/pith/LPWQEPXTNSKOFYG3BRZT7EV65U/action/author_attestation","sign_citation":"https://pith.science/pith/LPWQEPXTNSKOFYG3BRZT7EV65U/action/citation_signature","submit_replication":"https://pith.science/pith/LPWQEPXTNSKOFYG3BRZT7EV65U/action/replication_record"}},"created_at":"2026-07-05T08:29:18.491616+00:00","updated_at":"2026-07-05T08:29:18.491616+00:00"}