{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2017:XAR2G4OOHQ7DOBZZM5E2G6T2UY","short_pith_number":"pith:XAR2G4OO","schema_version":"1.0","canonical_sha256":"b823a371ce3c3e3707396749a37a7aa604e0961c31d6ba1e3e157132868724cd","source":{"kind":"arxiv","id":"1711.02827","version":2},"attestation_state":"computed","paper":{"title":"Inverse Reward Design","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Anca Dragan, Dylan Hadfield-Menell, Pieter Abbeel, Smitha Milli, Stuart Russell","submitted_at":"2017-11-08T04:44:32Z","abstract_excerpt":"Autonomous agents optimize the reward function we give them. What they don't know is how hard it is for us to design a reward function that actually captures what we want. When designing the reward, we might think of some specific training scenarios, and make sure that the reward will lead to the right behavior in those scenarios. Inevitably, agents encounter new scenarios (e.g., new types of terrain) where optimizing that same reward may lead to undesired behavior. Our insight is that reward functions are merely observations about what the designer actually wants, and that they should be inte"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1711.02827","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-11-08T04:44:32Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"5701965625a1fe7635eb804f6979b0e470eda9d483934f0c229f070347cba492","abstract_canon_sha256":"23de5188588901b042cd5be83263bb9f7d9422dcfed39e2e58fa43a0170af0b1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:41:05.669910Z","signature_b64":"Z7Pzi6e/cHbBK+v6qC3vshElbuO89Lzs74or3NsUAyMVAS2pDokapYBEDqVWVUdlrL1zPNPVi7xsiY49j9bPCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b823a371ce3c3e3707396749a37a7aa604e0961c31d6ba1e3e157132868724cd","last_reissued_at":"2026-07-05T01:41:05.669443Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:41:05.669443Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Inverse Reward Design","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Anca Dragan, Dylan Hadfield-Menell, Pieter Abbeel, Smitha Milli, Stuart Russell","submitted_at":"2017-11-08T04:44:32Z","abstract_excerpt":"Autonomous agents optimize the reward function we give them. What they don't know is how hard it is for us to design a reward function that actually captures what we want. When designing the reward, we might think of some specific training scenarios, and make sure that the reward will lead to the right behavior in those scenarios. Inevitably, agents encounter new scenarios (e.g., new types of terrain) where optimizing that same reward may lead to undesired behavior. Our insight is that reward functions are merely observations about what the designer actually wants, and that they should be inte"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1711.02827","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1711.02827/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1711.02827","created_at":"2026-07-05T01:41:05.669499+00:00"},{"alias_kind":"arxiv_version","alias_value":"1711.02827v2","created_at":"2026-07-05T01:41:05.669499+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1711.02827","created_at":"2026-07-05T01:41:05.669499+00:00"},{"alias_kind":"pith_short_12","alias_value":"XAR2G4OOHQ7D","created_at":"2026-07-05T01:41:05.669499+00:00"},{"alias_kind":"pith_short_16","alias_value":"XAR2G4OOHQ7DOBZZ","created_at":"2026-07-05T01:41:05.669499+00:00"},{"alias_kind":"pith_short_8","alias_value":"XAR2G4OO","created_at":"2026-07-05T01:41:05.669499+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.24155","citing_title":"The Alignment Target Problem: Divergent Moral Judgments of Humans, AI Systems, and Their Designers","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20265","citing_title":"Failure Modes of Maximum Entropy RLHF","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24155","citing_title":"The Alignment Target Problem: Divergent Moral Judgments of Humans, AI Systems, and Their Designers","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24155","citing_title":"The Alignment Target Problem: Divergent Moral Judgments of Humans, AI Systems, and Their Designers","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23210","citing_title":"Discovering Agentic Safety Specifications from 1-Bit Danger Signals","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY","json":"https://pith.science/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY.json","graph_json":"https://pith.science/api/pith-number/XAR2G4OOHQ7DOBZZM5E2G6T2UY/graph.json","events_json":"https://pith.science/api/pith-number/XAR2G4OOHQ7DOBZZM5E2G6T2UY/events.json","paper":"https://pith.science/paper/XAR2G4OO"},"agent_actions":{"view_html":"https://pith.science/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY","download_json":"https://pith.science/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY.json","view_paper":"https://pith.science/paper/XAR2G4OO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1711.02827&json=true","fetch_graph":"https://pith.science/api/pith-number/XAR2G4OOHQ7DOBZZM5E2G6T2UY/graph.json","fetch_events":"https://pith.science/api/pith-number/XAR2G4OOHQ7DOBZZM5E2G6T2UY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY/action/storage_attestation","attest_author":"https://pith.science/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY/action/author_attestation","sign_citation":"https://pith.science/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY/action/citation_signature","submit_replication":"https://pith.science/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY/action/replication_record"}},"created_at":"2026-07-05T01:41:05.669499+00:00","updated_at":"2026-07-05T01:41:05.669499+00:00"}