{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:O3E4ABA63FUC6KZXSBUQTNZZY5","short_pith_number":"pith:O3E4ABA6","schema_version":"1.0","canonical_sha256":"76c9c0041ed9682f2b37906909b739c7748bbf0dd1d166e4c2b1256362addae1","source":{"kind":"arxiv","id":"1909.05477","version":2},"attestation_state":"computed","paper":{"title":"Maximum Likelihood Constraint Inference for Inverse Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO","cs.SY","eess.SY","stat.ML"],"primary_cat":"cs.LG","authors_text":"Dexter R.R. Scobee, S. Shankar Sastry","submitted_at":"2019-09-12T06:38:46Z","abstract_excerpt":"While most approaches to the problem of Inverse Reinforcement Learning (IRL) focus on estimating a reward function that best explains an expert agent's policy or demonstrated behavior on a control task, it is often the case that such behavior is more succinctly represented by a simple reward combined with a set of hard constraints. In this setting, the agent is attempting to maximize cumulative rewards subject to these given constraints on their behavior. We reformulate the problem of IRL on Markov Decision Processes (MDPs) such that, given a nominal model of the environment and a nominal rewa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1909.05477","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-09-12T06:38:46Z","cross_cats_sorted":["cs.AI","cs.RO","cs.SY","eess.SY","stat.ML"],"title_canon_sha256":"a7af58814701fcc6a5a5544a1f216c1ab656591a946a55bea8ab33ba638678d7","abstract_canon_sha256":"8e7e6213c27046ed912af488ac99c3c6b3851b4b49c28a652dfd1a1775351684"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:59:22.557288Z","signature_b64":"skDTYdfoM3U2RvHYiGigS069KZi6njTnRGoXQA+XAFSa4BDnNRRgbKlUXZ3PtqcsimJN6PnSb8mxEeHJ0pCHCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"76c9c0041ed9682f2b37906909b739c7748bbf0dd1d166e4c2b1256362addae1","last_reissued_at":"2026-07-05T00:59:22.556864Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:59:22.556864Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Maximum Likelihood Constraint Inference for Inverse Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO","cs.SY","eess.SY","stat.ML"],"primary_cat":"cs.LG","authors_text":"Dexter R.R. Scobee, S. Shankar Sastry","submitted_at":"2019-09-12T06:38:46Z","abstract_excerpt":"While most approaches to the problem of Inverse Reinforcement Learning (IRL) focus on estimating a reward function that best explains an expert agent's policy or demonstrated behavior on a control task, it is often the case that such behavior is more succinctly represented by a simple reward combined with a set of hard constraints. In this setting, the agent is attempting to maximize cumulative rewards subject to these given constraints on their behavior. We reformulate the problem of IRL on Markov Decision Processes (MDPs) such that, given a nominal model of the environment and a nominal rewa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1909.05477","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1909.05477/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1909.05477","created_at":"2026-07-05T00:59:22.556920+00:00"},{"alias_kind":"arxiv_version","alias_value":"1909.05477v2","created_at":"2026-07-05T00:59:22.556920+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1909.05477","created_at":"2026-07-05T00:59:22.556920+00:00"},{"alias_kind":"pith_short_12","alias_value":"O3E4ABA63FUC","created_at":"2026-07-05T00:59:22.556920+00:00"},{"alias_kind":"pith_short_16","alias_value":"O3E4ABA63FUC6KZX","created_at":"2026-07-05T00:59:22.556920+00:00"},{"alias_kind":"pith_short_8","alias_value":"O3E4ABA6","created_at":"2026-07-05T00:59:22.556920+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2411.05174","citing_title":"Bayesian Inverse Transition Learning: Learning Dynamics From Near-Optimal Trajectories","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06951","citing_title":"Multi-Objective Constraint Inference using Inverse reinforcement learning","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O3E4ABA63FUC6KZXSBUQTNZZY5","json":"https://pith.science/pith/O3E4ABA63FUC6KZXSBUQTNZZY5.json","graph_json":"https://pith.science/api/pith-number/O3E4ABA63FUC6KZXSBUQTNZZY5/graph.json","events_json":"https://pith.science/api/pith-number/O3E4ABA63FUC6KZXSBUQTNZZY5/events.json","paper":"https://pith.science/paper/O3E4ABA6"},"agent_actions":{"view_html":"https://pith.science/pith/O3E4ABA63FUC6KZXSBUQTNZZY5","download_json":"https://pith.science/pith/O3E4ABA63FUC6KZXSBUQTNZZY5.json","view_paper":"https://pith.science/paper/O3E4ABA6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1909.05477&json=true","fetch_graph":"https://pith.science/api/pith-number/O3E4ABA63FUC6KZXSBUQTNZZY5/graph.json","fetch_events":"https://pith.science/api/pith-number/O3E4ABA63FUC6KZXSBUQTNZZY5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O3E4ABA63FUC6KZXSBUQTNZZY5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O3E4ABA63FUC6KZXSBUQTNZZY5/action/storage_attestation","attest_author":"https://pith.science/pith/O3E4ABA63FUC6KZXSBUQTNZZY5/action/author_attestation","sign_citation":"https://pith.science/pith/O3E4ABA63FUC6KZXSBUQTNZZY5/action/citation_signature","submit_replication":"https://pith.science/pith/O3E4ABA63FUC6KZXSBUQTNZZY5/action/replication_record"}},"created_at":"2026-07-05T00:59:22.556920+00:00","updated_at":"2026-07-05T00:59:22.556920+00:00"}