{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:RNN57ZE537UXL2XY2A4H4RKYTN","short_pith_number":"pith:RNN57ZE5","schema_version":"1.0","canonical_sha256":"8b5bdfe49ddfe975eaf8d0387e45589b4828b71577df74331060f42e579b51ba","source":{"kind":"arxiv","id":"2212.03201","version":2},"attestation_state":"computed","paper":{"title":"Misspecification in Inverse Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alessandro Abate, Joar Skalse","submitted_at":"2022-12-06T18:21:47Z","abstract_excerpt":"The aim of Inverse Reinforcement Learning (IRL) is to infer a reward function $R$ from a policy $\\pi$. To do this, we need a model of how $\\pi$ relates to $R$. In the current literature, the most common models are optimality, Boltzmann rationality, and causal entropy maximisation. One of the primary motivations behind IRL is to infer human preferences from human behaviour. However, the true relationship between human preferences and human behaviour is much more complex than any of the models currently used in IRL. This means that they are misspecified, which raises the worry that they might le"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.03201","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-12-06T18:21:47Z","cross_cats_sorted":[],"title_canon_sha256":"f8f616701f615651c95144be1bb86a558e3114e19b89fc0457744e6b4aa91e60","abstract_canon_sha256":"0f4e33e0c093c37f70da37d6517aa718ee6c88e0d57861c42faf241c5d9cbf10"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:54:14.354710Z","signature_b64":"cQ5UpLoonxlJpCx1bvO2ca9sNEvr+1YDKeZM09zo5WvlLq8XKzy4zudU8FMFQvw1q43CbXbMMPoDCAalppEIAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b5bdfe49ddfe975eaf8d0387e45589b4828b71577df74331060f42e579b51ba","last_reissued_at":"2026-07-05T05:54:14.354164Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:54:14.354164Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Misspecification in Inverse Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alessandro Abate, Joar Skalse","submitted_at":"2022-12-06T18:21:47Z","abstract_excerpt":"The aim of Inverse Reinforcement Learning (IRL) is to infer a reward function $R$ from a policy $\\pi$. To do this, we need a model of how $\\pi$ relates to $R$. In the current literature, the most common models are optimality, Boltzmann rationality, and causal entropy maximisation. One of the primary motivations behind IRL is to infer human preferences from human behaviour. However, the true relationship between human preferences and human behaviour is much more complex than any of the models currently used in IRL. This means that they are misspecified, which raises the worry that they might le"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.03201","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.03201/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.03201","created_at":"2026-07-05T05:54:14.354229+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.03201v2","created_at":"2026-07-05T05:54:14.354229+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.03201","created_at":"2026-07-05T05:54:14.354229+00:00"},{"alias_kind":"pith_short_12","alias_value":"RNN57ZE537UX","created_at":"2026-07-05T05:54:14.354229+00:00"},{"alias_kind":"pith_short_16","alias_value":"RNN57ZE537UXL2XY","created_at":"2026-07-05T05:54:14.354229+00:00"},{"alias_kind":"pith_short_8","alias_value":"RNN57ZE5","created_at":"2026-07-05T05:54:14.354229+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2310.15288","citing_title":"Active teacher selection for reward learning","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RNN57ZE537UXL2XY2A4H4RKYTN","json":"https://pith.science/pith/RNN57ZE537UXL2XY2A4H4RKYTN.json","graph_json":"https://pith.science/api/pith-number/RNN57ZE537UXL2XY2A4H4RKYTN/graph.json","events_json":"https://pith.science/api/pith-number/RNN57ZE537UXL2XY2A4H4RKYTN/events.json","paper":"https://pith.science/paper/RNN57ZE5"},"agent_actions":{"view_html":"https://pith.science/pith/RNN57ZE537UXL2XY2A4H4RKYTN","download_json":"https://pith.science/pith/RNN57ZE537UXL2XY2A4H4RKYTN.json","view_paper":"https://pith.science/paper/RNN57ZE5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.03201&json=true","fetch_graph":"https://pith.science/api/pith-number/RNN57ZE537UXL2XY2A4H4RKYTN/graph.json","fetch_events":"https://pith.science/api/pith-number/RNN57ZE537UXL2XY2A4H4RKYTN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RNN57ZE537UXL2XY2A4H4RKYTN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RNN57ZE537UXL2XY2A4H4RKYTN/action/storage_attestation","attest_author":"https://pith.science/pith/RNN57ZE537UXL2XY2A4H4RKYTN/action/author_attestation","sign_citation":"https://pith.science/pith/RNN57ZE537UXL2XY2A4H4RKYTN/action/citation_signature","submit_replication":"https://pith.science/pith/RNN57ZE537UXL2XY2A4H4RKYTN/action/replication_record"}},"created_at":"2026-07-05T05:54:14.354229+00:00","updated_at":"2026-07-05T05:54:14.354229+00:00"}