{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:7GPT2T2W76GPV4B6QX6XSYWIP6","short_pith_number":"pith:7GPT2T2W","schema_version":"1.0","canonical_sha256":"f99f3d4f56ff8cfaf03e85fd7962c87f9f40554b1fa5494916cdf66cb0355079","source":{"kind":"arxiv","id":"2302.07457","version":3},"attestation_state":"computed","paper":{"title":"When Demonstrations Meet Generative World Models: A Maximum Likelihood Framework for Offline Inverse Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alfredo Garcia, Chenliang Li, Mingyi Hong, Siliang Zeng","submitted_at":"2023-02-15T04:14:20Z","abstract_excerpt":"Offline inverse reinforcement learning (Offline IRL) aims to recover the structure of rewards and environment dynamics that underlie observed actions in a fixed, finite set of demonstrations from an expert agent. Accurate models of expertise in executing a task has applications in safety-sensitive applications such as clinical decision making and autonomous driving. However, the structure of an expert's preferences implicit in observed actions is closely linked to the expert's model of the environment dynamics (i.e. the ``world'' model). Thus, inaccurate models of the world obtained from finit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.07457","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-02-15T04:14:20Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7716ffab57c72ae65d85bb57cb06965cca5848737d01eb488857a513ce00f7b6","abstract_canon_sha256":"51a4f185293ece5bd97a15886cbfcdc6b4def8fbe465e75996133ba0125b8025"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:50:18.993993Z","signature_b64":"tk66uZJ1pI5wplJSg/rAPheSYMySNvIWk4ef7lWJfl9otbqSTLSc5VxouppSw7udIkS2F2Opsv/Ujasy4WPDCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f99f3d4f56ff8cfaf03e85fd7962c87f9f40554b1fa5494916cdf66cb0355079","last_reissued_at":"2026-07-05T07:50:18.993568Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:50:18.993568Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Demonstrations Meet Generative World Models: A Maximum Likelihood Framework for Offline Inverse Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alfredo Garcia, Chenliang Li, Mingyi Hong, Siliang Zeng","submitted_at":"2023-02-15T04:14:20Z","abstract_excerpt":"Offline inverse reinforcement learning (Offline IRL) aims to recover the structure of rewards and environment dynamics that underlie observed actions in a fixed, finite set of demonstrations from an expert agent. Accurate models of expertise in executing a task has applications in safety-sensitive applications such as clinical decision making and autonomous driving. However, the structure of an expert's preferences implicit in observed actions is closely linked to the expert's model of the environment dynamics (i.e. the ``world'' model). Thus, inaccurate models of the world obtained from finit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.07457","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.07457/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.07457","created_at":"2026-07-05T07:50:18.993628+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.07457v3","created_at":"2026-07-05T07:50:18.993628+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.07457","created_at":"2026-07-05T07:50:18.993628+00:00"},{"alias_kind":"pith_short_12","alias_value":"7GPT2T2W76GP","created_at":"2026-07-05T07:50:18.993628+00:00"},{"alias_kind":"pith_short_16","alias_value":"7GPT2T2W76GPV4B6","created_at":"2026-07-05T07:50:18.993628+00:00"},{"alias_kind":"pith_short_8","alias_value":"7GPT2T2W","created_at":"2026-07-05T07:50:18.993628+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7GPT2T2W76GPV4B6QX6XSYWIP6","json":"https://pith.science/pith/7GPT2T2W76GPV4B6QX6XSYWIP6.json","graph_json":"https://pith.science/api/pith-number/7GPT2T2W76GPV4B6QX6XSYWIP6/graph.json","events_json":"https://pith.science/api/pith-number/7GPT2T2W76GPV4B6QX6XSYWIP6/events.json","paper":"https://pith.science/paper/7GPT2T2W"},"agent_actions":{"view_html":"https://pith.science/pith/7GPT2T2W76GPV4B6QX6XSYWIP6","download_json":"https://pith.science/pith/7GPT2T2W76GPV4B6QX6XSYWIP6.json","view_paper":"https://pith.science/paper/7GPT2T2W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.07457&json=true","fetch_graph":"https://pith.science/api/pith-number/7GPT2T2W76GPV4B6QX6XSYWIP6/graph.json","fetch_events":"https://pith.science/api/pith-number/7GPT2T2W76GPV4B6QX6XSYWIP6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7GPT2T2W76GPV4B6QX6XSYWIP6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7GPT2T2W76GPV4B6QX6XSYWIP6/action/storage_attestation","attest_author":"https://pith.science/pith/7GPT2T2W76GPV4B6QX6XSYWIP6/action/author_attestation","sign_citation":"https://pith.science/pith/7GPT2T2W76GPV4B6QX6XSYWIP6/action/citation_signature","submit_replication":"https://pith.science/pith/7GPT2T2W76GPV4B6QX6XSYWIP6/action/replication_record"}},"created_at":"2026-07-05T07:50:18.993628+00:00","updated_at":"2026-07-05T07:50:18.993628+00:00"}