{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:65MPQ73Z7YNOV3D4AKXGX2XGDM","short_pith_number":"pith:65MPQ73Z","canonical_record":{"source":{"id":"2111.06956","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-11-12T21:44:15Z","cross_cats_sorted":[],"title_canon_sha256":"8d153223d6ae43a773325d9688c08049c9d7e112e1b726af8d672a37afc5eeb3","abstract_canon_sha256":"173472dd1b77dfe998c9cf2b65a85c68a03bd7ae43859e6e829566da8fe989c7"},"schema_version":"1.0"},"canonical_sha256":"f758f87f79fe1aeaec7c02ae6beae61b36a273f8e2c99281b4d78f81136f9685","source":{"kind":"arxiv","id":"2111.06956","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2111.06956","created_at":"2026-07-05T03:31:33Z"},{"alias_kind":"arxiv_version","alias_value":"2111.06956v1","created_at":"2026-07-05T03:31:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.06956","created_at":"2026-07-05T03:31:33Z"},{"alias_kind":"pith_short_12","alias_value":"65MPQ73Z7YNO","created_at":"2026-07-05T03:31:33Z"},{"alias_kind":"pith_short_16","alias_value":"65MPQ73Z7YNOV3D4","created_at":"2026-07-05T03:31:33Z"},{"alias_kind":"pith_short_8","alias_value":"65MPQ73Z","created_at":"2026-07-05T03:31:33Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:65MPQ73Z7YNOV3D4AKXGX2XGDM","target":"record","payload":{"canonical_record":{"source":{"id":"2111.06956","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-11-12T21:44:15Z","cross_cats_sorted":[],"title_canon_sha256":"8d153223d6ae43a773325d9688c08049c9d7e112e1b726af8d672a37afc5eeb3","abstract_canon_sha256":"173472dd1b77dfe998c9cf2b65a85c68a03bd7ae43859e6e829566da8fe989c7"},"schema_version":"1.0"},"canonical_sha256":"f758f87f79fe1aeaec7c02ae6beae61b36a273f8e2c99281b4d78f81136f9685","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:31:33.608354Z","signature_b64":"8WgBuEjbqqD/35AJsJuhe3nESIQcnlEZB1S+Qk7Fpzht+Bsu/JzkBvdDRxBRjIqpPYzM0IP/X4HFhIvBeG/uAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f758f87f79fe1aeaec7c02ae6beae61b36a273f8e2c99281b4d78f81136f9685","last_reissued_at":"2026-07-05T03:31:33.607913Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:31:33.607913Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2111.06956","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:31:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"PfNnkvGF5BIU7O7ZmiNLMXwyHeLX7QeHNrGA9RRcN0YyESa3I7Ns09izeteXTb5zh58X9PfvVcr1ZjOvlHd7AQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T13:28:55.356785Z"},"content_sha256":"960cdb185dd8bd2c83ebd30bcc996a1856dd391ab6614df59183bd98e7b4d177","schema_version":"1.0","event_id":"sha256:960cdb185dd8bd2c83ebd30bcc996a1856dd391ab6614df59183bd98e7b4d177"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:65MPQ73Z7YNOV3D4AKXGX2XGDM","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Human irrationality: both bad and good for reward inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Anca Dragan, Andrew Critch, Lawrence Chan","submitted_at":"2021-11-12T21:44:15Z","abstract_excerpt":"Assuming humans are (approximately) rational enables robots to infer reward functions by observing human behavior. But people exhibit a wide array of irrationalities, and our goal with this work is to better understand the effect they can have on reward inference. The challenge with studying this effect is that there are many types of irrationality, with varying degrees of mathematical formalization. We thus operationalize irrationality in the language of MDPs, by altering the Bellman optimality equation, and use this framework to study how these alterations would affect inference.\n  We find t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.06956","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.06956/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:31:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"JDAa3CiJJ0VUxr6mnfr3nrMBp9QueQzwT9IlquCwP0202r1akSBZvWHPuZPZPnB3sXLgC/vcnFDNI/k+jSgOAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T13:28:55.357347Z"},"content_sha256":"d4e250ce05c78501ed8d07b7a05c5ab6f82421c5dc82fe90e9e139e9470186d7","schema_version":"1.0","event_id":"sha256:d4e250ce05c78501ed8d07b7a05c5ab6f82421c5dc82fe90e9e139e9470186d7"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/65MPQ73Z7YNOV3D4AKXGX2XGDM/bundle.json","state_url":"https://pith.science/pith/65MPQ73Z7YNOV3D4AKXGX2XGDM/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/65MPQ73Z7YNOV3D4AKXGX2XGDM/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T13:28:55Z","links":{"resolver":"https://pith.science/pith/65MPQ73Z7YNOV3D4AKXGX2XGDM","bundle":"https://pith.science/pith/65MPQ73Z7YNOV3D4AKXGX2XGDM/bundle.json","state":"https://pith.science/pith/65MPQ73Z7YNOV3D4AKXGX2XGDM/state.json","well_known_bundle":"https://pith.science/.well-known/pith/65MPQ73Z7YNOV3D4AKXGX2XGDM/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:65MPQ73Z7YNOV3D4AKXGX2XGDM","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"173472dd1b77dfe998c9cf2b65a85c68a03bd7ae43859e6e829566da8fe989c7","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-11-12T21:44:15Z","title_canon_sha256":"8d153223d6ae43a773325d9688c08049c9d7e112e1b726af8d672a37afc5eeb3"},"schema_version":"1.0","source":{"id":"2111.06956","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2111.06956","created_at":"2026-07-05T03:31:33Z"},{"alias_kind":"arxiv_version","alias_value":"2111.06956v1","created_at":"2026-07-05T03:31:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.06956","created_at":"2026-07-05T03:31:33Z"},{"alias_kind":"pith_short_12","alias_value":"65MPQ73Z7YNO","created_at":"2026-07-05T03:31:33Z"},{"alias_kind":"pith_short_16","alias_value":"65MPQ73Z7YNOV3D4","created_at":"2026-07-05T03:31:33Z"},{"alias_kind":"pith_short_8","alias_value":"65MPQ73Z","created_at":"2026-07-05T03:31:33Z"}],"graph_snapshots":[{"event_id":"sha256:d4e250ce05c78501ed8d07b7a05c5ab6f82421c5dc82fe90e9e139e9470186d7","target":"graph","created_at":"2026-07-05T03:31:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2111.06956/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Assuming humans are (approximately) rational enables robots to infer reward functions by observing human behavior. But people exhibit a wide array of irrationalities, and our goal with this work is to better understand the effect they can have on reward inference. The challenge with studying this effect is that there are many types of irrationality, with varying degrees of mathematical formalization. We thus operationalize irrationality in the language of MDPs, by altering the Bellman optimality equation, and use this framework to study how these alterations would affect inference.\n  We find t","authors_text":"Anca Dragan, Andrew Critch, Lawrence Chan","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-11-12T21:44:15Z","title":"Human irrationality: both bad and good for reward inference"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.06956","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:960cdb185dd8bd2c83ebd30bcc996a1856dd391ab6614df59183bd98e7b4d177","target":"record","created_at":"2026-07-05T03:31:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"173472dd1b77dfe998c9cf2b65a85c68a03bd7ae43859e6e829566da8fe989c7","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-11-12T21:44:15Z","title_canon_sha256":"8d153223d6ae43a773325d9688c08049c9d7e112e1b726af8d672a37afc5eeb3"},"schema_version":"1.0","source":{"id":"2111.06956","kind":"arxiv","version":1}},"canonical_sha256":"f758f87f79fe1aeaec7c02ae6beae61b36a273f8e2c99281b4d78f81136f9685","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"f758f87f79fe1aeaec7c02ae6beae61b36a273f8e2c99281b4d78f81136f9685","first_computed_at":"2026-07-05T03:31:33.607913Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:31:33.607913Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"8WgBuEjbqqD/35AJsJuhe3nESIQcnlEZB1S+Qk7Fpzht+Bsu/JzkBvdDRxBRjIqpPYzM0IP/X4HFhIvBeG/uAw==","signature_status":"signed_v1","signed_at":"2026-07-05T03:31:33.608354Z","signed_message":"canonical_sha256_bytes"},"source_id":"2111.06956","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:960cdb185dd8bd2c83ebd30bcc996a1856dd391ab6614df59183bd98e7b4d177","sha256:d4e250ce05c78501ed8d07b7a05c5ab6f82421c5dc82fe90e9e139e9470186d7"],"state_sha256":"4b8ec7b7222208a43bca07f97fbefaa91725cfe149d32cba6ce3dd8a7aaee447"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"XW8tpsFTELTbmckeIk4RqyXA+LHKVHB8qJwWYTyCOALYOy8ww+QiUx5YkzXPNKpFQGMDBCZal50Qv/c5yBhvBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T13:28:55.360752Z","bundle_sha256":"36dfe222ec60e849cdb1d03572168582d951a5702bdab45e1aa71437cf4ea4b3"}}