{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:4YE7E5IJQTFMFUIDZVKKAWUMNZ","short_pith_number":"pith:4YE7E5IJ","schema_version":"1.0","canonical_sha256":"e609f2750984cac2d103cd54a05a8c6e61720a24e1c164526ef046a9a5628424","source":{"kind":"arxiv","id":"2312.00054","version":2},"attestation_state":"computed","paper":{"title":"Is Inverse Reinforcement Learning Harder than Standard Reinforcement Learning? A Theoretical Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"stat.ML","authors_text":"Lei Zhao, Mengdi Wang, Yu Bai","submitted_at":"2023-11-29T00:09:01Z","abstract_excerpt":"Inverse Reinforcement Learning (IRL) -- the problem of learning reward functions from demonstrations of an \\emph{expert policy} -- plays a critical role in developing intelligent systems. While widely used in applications, theoretical understandings of IRL present unique challenges and remain less developed compared with standard RL. For example, it remains open how to do IRL efficiently in standard \\emph{offline} settings with pre-collected data, where states are obtained from a \\emph{behavior policy} (which could be the expert policy itself), and actions are sampled from the expert policy.\n "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.00054","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2023-11-29T00:09:01Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"cf58d181bf7e0ec40ba380a1513feb303b1f5c3e9de1d830cd887e0ba63ba178","abstract_canon_sha256":"e55f509fb9b56994c0270a1130cd4171bc8a3c6d46a8781b99ee40a0b3701488"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:43:34.301656Z","signature_b64":"wMpzAx1r38V+L0f/S/vEdnX1JxQJQ3SbUfNoniMu6mRrBX90RZ1cvOwZ+611oOz4TirIIB+5i+st7wAG/9K2Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e609f2750984cac2d103cd54a05a8c6e61720a24e1c164526ef046a9a5628424","last_reissued_at":"2026-07-05T07:43:34.301133Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:43:34.301133Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Is Inverse Reinforcement Learning Harder than Standard Reinforcement Learning? A Theoretical Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"stat.ML","authors_text":"Lei Zhao, Mengdi Wang, Yu Bai","submitted_at":"2023-11-29T00:09:01Z","abstract_excerpt":"Inverse Reinforcement Learning (IRL) -- the problem of learning reward functions from demonstrations of an \\emph{expert policy} -- plays a critical role in developing intelligent systems. While widely used in applications, theoretical understandings of IRL present unique challenges and remain less developed compared with standard RL. For example, it remains open how to do IRL efficiently in standard \\emph{offline} settings with pre-collected data, where states are obtained from a \\emph{behavior policy} (which could be the expert policy itself), and actions are sampled from the expert policy.\n "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.00054","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.00054/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.00054","created_at":"2026-07-05T07:43:34.301188+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.00054v2","created_at":"2026-07-05T07:43:34.301188+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.00054","created_at":"2026-07-05T07:43:34.301188+00:00"},{"alias_kind":"pith_short_12","alias_value":"4YE7E5IJQTFM","created_at":"2026-07-05T07:43:34.301188+00:00"},{"alias_kind":"pith_short_16","alias_value":"4YE7E5IJQTFMFUID","created_at":"2026-07-05T07:43:34.301188+00:00"},{"alias_kind":"pith_short_8","alias_value":"4YE7E5IJ","created_at":"2026-07-05T07:43:34.301188+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.17249","citing_title":"Where You Go is Who You Are: Behavioral Theory-Guided LLMs for Inverse Reinforcement Learning","ref_index":57,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4YE7E5IJQTFMFUIDZVKKAWUMNZ","json":"https://pith.science/pith/4YE7E5IJQTFMFUIDZVKKAWUMNZ.json","graph_json":"https://pith.science/api/pith-number/4YE7E5IJQTFMFUIDZVKKAWUMNZ/graph.json","events_json":"https://pith.science/api/pith-number/4YE7E5IJQTFMFUIDZVKKAWUMNZ/events.json","paper":"https://pith.science/paper/4YE7E5IJ"},"agent_actions":{"view_html":"https://pith.science/pith/4YE7E5IJQTFMFUIDZVKKAWUMNZ","download_json":"https://pith.science/pith/4YE7E5IJQTFMFUIDZVKKAWUMNZ.json","view_paper":"https://pith.science/paper/4YE7E5IJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.00054&json=true","fetch_graph":"https://pith.science/api/pith-number/4YE7E5IJQTFMFUIDZVKKAWUMNZ/graph.json","fetch_events":"https://pith.science/api/pith-number/4YE7E5IJQTFMFUIDZVKKAWUMNZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4YE7E5IJQTFMFUIDZVKKAWUMNZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4YE7E5IJQTFMFUIDZVKKAWUMNZ/action/storage_attestation","attest_author":"https://pith.science/pith/4YE7E5IJQTFMFUIDZVKKAWUMNZ/action/author_attestation","sign_citation":"https://pith.science/pith/4YE7E5IJQTFMFUIDZVKKAWUMNZ/action/citation_signature","submit_replication":"https://pith.science/pith/4YE7E5IJQTFMFUIDZVKKAWUMNZ/action/replication_record"}},"created_at":"2026-07-05T07:43:34.301188+00:00","updated_at":"2026-07-05T07:43:34.301188+00:00"}