{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2017:XAR2G4OOHQ7DOBZZM5E2G6T2UY","short_pith_number":"pith:XAR2G4OO","canonical_record":{"source":{"id":"1711.02827","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-11-08T04:44:32Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"5701965625a1fe7635eb804f6979b0e470eda9d483934f0c229f070347cba492","abstract_canon_sha256":"23de5188588901b042cd5be83263bb9f7d9422dcfed39e2e58fa43a0170af0b1"},"schema_version":"1.0"},"canonical_sha256":"b823a371ce3c3e3707396749a37a7aa604e0961c31d6ba1e3e157132868724cd","source":{"kind":"arxiv","id":"1711.02827","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1711.02827","created_at":"2026-07-05T01:41:05Z"},{"alias_kind":"arxiv_version","alias_value":"1711.02827v2","created_at":"2026-07-05T01:41:05Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1711.02827","created_at":"2026-07-05T01:41:05Z"},{"alias_kind":"pith_short_12","alias_value":"XAR2G4OOHQ7D","created_at":"2026-07-05T01:41:05Z"},{"alias_kind":"pith_short_16","alias_value":"XAR2G4OOHQ7DOBZZ","created_at":"2026-07-05T01:41:05Z"},{"alias_kind":"pith_short_8","alias_value":"XAR2G4OO","created_at":"2026-07-05T01:41:05Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2017:XAR2G4OOHQ7DOBZZM5E2G6T2UY","target":"record","payload":{"canonical_record":{"source":{"id":"1711.02827","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-11-08T04:44:32Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"5701965625a1fe7635eb804f6979b0e470eda9d483934f0c229f070347cba492","abstract_canon_sha256":"23de5188588901b042cd5be83263bb9f7d9422dcfed39e2e58fa43a0170af0b1"},"schema_version":"1.0"},"canonical_sha256":"b823a371ce3c3e3707396749a37a7aa604e0961c31d6ba1e3e157132868724cd","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:41:05.669910Z","signature_b64":"Z7Pzi6e/cHbBK+v6qC3vshElbuO89Lzs74or3NsUAyMVAS2pDokapYBEDqVWVUdlrL1zPNPVi7xsiY49j9bPCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b823a371ce3c3e3707396749a37a7aa604e0961c31d6ba1e3e157132868724cd","last_reissued_at":"2026-07-05T01:41:05.669443Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:41:05.669443Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1711.02827","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T01:41:05Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"MCMenZXh95GKakS/wvJvjO9Zl6Jk+gOV0WG0AhGeW2l4YlvIAhOll+zCbxW+V/yi07sL+m3oxfrX+hpd3A/BAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T19:32:36.939181Z"},"content_sha256":"7360c10021fcf290203463995b9d2a324767e43873a3d82d8c33c49c8c3ed88a","schema_version":"1.0","event_id":"sha256:7360c10021fcf290203463995b9d2a324767e43873a3d82d8c33c49c8c3ed88a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2017:XAR2G4OOHQ7DOBZZM5E2G6T2UY","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Inverse Reward Design","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Anca Dragan, Dylan Hadfield-Menell, Pieter Abbeel, Smitha Milli, Stuart Russell","submitted_at":"2017-11-08T04:44:32Z","abstract_excerpt":"Autonomous agents optimize the reward function we give them. What they don't know is how hard it is for us to design a reward function that actually captures what we want. When designing the reward, we might think of some specific training scenarios, and make sure that the reward will lead to the right behavior in those scenarios. Inevitably, agents encounter new scenarios (e.g., new types of terrain) where optimizing that same reward may lead to undesired behavior. Our insight is that reward functions are merely observations about what the designer actually wants, and that they should be inte"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1711.02827","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1711.02827/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T01:41:05Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"wMK0/lW3/I3GLx2DNlarmr2DI1mzREJLy8YrNk5NJFFKelq7qAzMjDeRtLMkUlQViYIDPaad1kQhqlkDiP1+Dg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T19:32:36.939676Z"},"content_sha256":"20ac88fefc9c757ad62254aed50a6a6eeda6dc19709ba82af74e4fd927fee0c8","schema_version":"1.0","event_id":"sha256:20ac88fefc9c757ad62254aed50a6a6eeda6dc19709ba82af74e4fd927fee0c8"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY/bundle.json","state_url":"https://pith.science/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T19:32:36Z","links":{"resolver":"https://pith.science/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY","bundle":"https://pith.science/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY/bundle.json","state":"https://pith.science/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY/state.json","well_known_bundle":"https://pith.science/.well-known/pith/XAR2G4OOHQ7DOBZZM5E2G6T2UY/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2017:XAR2G4OOHQ7DOBZZM5E2G6T2UY","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"23de5188588901b042cd5be83263bb9f7d9422dcfed39e2e58fa43a0170af0b1","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-11-08T04:44:32Z","title_canon_sha256":"5701965625a1fe7635eb804f6979b0e470eda9d483934f0c229f070347cba492"},"schema_version":"1.0","source":{"id":"1711.02827","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1711.02827","created_at":"2026-07-05T01:41:05Z"},{"alias_kind":"arxiv_version","alias_value":"1711.02827v2","created_at":"2026-07-05T01:41:05Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1711.02827","created_at":"2026-07-05T01:41:05Z"},{"alias_kind":"pith_short_12","alias_value":"XAR2G4OOHQ7D","created_at":"2026-07-05T01:41:05Z"},{"alias_kind":"pith_short_16","alias_value":"XAR2G4OOHQ7DOBZZ","created_at":"2026-07-05T01:41:05Z"},{"alias_kind":"pith_short_8","alias_value":"XAR2G4OO","created_at":"2026-07-05T01:41:05Z"}],"graph_snapshots":[{"event_id":"sha256:20ac88fefc9c757ad62254aed50a6a6eeda6dc19709ba82af74e4fd927fee0c8","target":"graph","created_at":"2026-07-05T01:41:05Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/1711.02827/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Autonomous agents optimize the reward function we give them. What they don't know is how hard it is for us to design a reward function that actually captures what we want. When designing the reward, we might think of some specific training scenarios, and make sure that the reward will lead to the right behavior in those scenarios. Inevitably, agents encounter new scenarios (e.g., new types of terrain) where optimizing that same reward may lead to undesired behavior. Our insight is that reward functions are merely observations about what the designer actually wants, and that they should be inte","authors_text":"Anca Dragan, Dylan Hadfield-Menell, Pieter Abbeel, Smitha Milli, Stuart Russell","cross_cats":["cs.LG"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-11-08T04:44:32Z","title":"Inverse Reward Design"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1711.02827","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:7360c10021fcf290203463995b9d2a324767e43873a3d82d8c33c49c8c3ed88a","target":"record","created_at":"2026-07-05T01:41:05Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"23de5188588901b042cd5be83263bb9f7d9422dcfed39e2e58fa43a0170af0b1","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-11-08T04:44:32Z","title_canon_sha256":"5701965625a1fe7635eb804f6979b0e470eda9d483934f0c229f070347cba492"},"schema_version":"1.0","source":{"id":"1711.02827","kind":"arxiv","version":2}},"canonical_sha256":"b823a371ce3c3e3707396749a37a7aa604e0961c31d6ba1e3e157132868724cd","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b823a371ce3c3e3707396749a37a7aa604e0961c31d6ba1e3e157132868724cd","first_computed_at":"2026-07-05T01:41:05.669443Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T01:41:05.669443Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Z7Pzi6e/cHbBK+v6qC3vshElbuO89Lzs74or3NsUAyMVAS2pDokapYBEDqVWVUdlrL1zPNPVi7xsiY49j9bPCw==","signature_status":"signed_v1","signed_at":"2026-07-05T01:41:05.669910Z","signed_message":"canonical_sha256_bytes"},"source_id":"1711.02827","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:7360c10021fcf290203463995b9d2a324767e43873a3d82d8c33c49c8c3ed88a","sha256:20ac88fefc9c757ad62254aed50a6a6eeda6dc19709ba82af74e4fd927fee0c8"],"state_sha256":"e033f3ac2c92186616bcfa823524aa89be9ec075e949f62d932851857a88b8ef"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"XZvL86xVIWE5hizk6zSTLXm40zlWxmYmX83wQ7vA3kyc6fH/K0aYUxaFwbSlvgpr3zhJGx+NsA8tcQFe7IoAAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T19:32:36.943718Z","bundle_sha256":"8924a420d7d683b314edd9714f0f01f898ef74ded70a15028153c03b89595dc2"}}