{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:3V2FUOVGLKAJUJ42Z5XLFPFSYU","short_pith_number":"pith:3V2FUOVG","canonical_record":{"source":{"id":"2206.02231","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-06-05T17:58:02Z","cross_cats_sorted":["cs.AI","cs.SY","eess.SY"],"title_canon_sha256":"d82b36090e1b884c71f2701820d7024483dc079f2399abba4172f6175fa0021c","abstract_canon_sha256":"fdb2a99e088a783203be25b87a1227d696195f30168f4dfcf3429c355272874b"},"schema_version":"1.0"},"canonical_sha256":"dd745a3aa65a809a279acf6eb2bcb2c52256d8db190c38cf1a1eb000c3dc4f8d","source":{"kind":"arxiv","id":"2206.02231","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2206.02231","created_at":"2026-07-05T06:48:22Z"},{"alias_kind":"arxiv_version","alias_value":"2206.02231v3","created_at":"2026-07-05T06:48:22Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.02231","created_at":"2026-07-05T06:48:22Z"},{"alias_kind":"pith_short_12","alias_value":"3V2FUOVGLKAJ","created_at":"2026-07-05T06:48:22Z"},{"alias_kind":"pith_short_16","alias_value":"3V2FUOVGLKAJUJ42","created_at":"2026-07-05T06:48:22Z"},{"alias_kind":"pith_short_8","alias_value":"3V2FUOVG","created_at":"2026-07-05T06:48:22Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:3V2FUOVGLKAJUJ42Z5XLFPFSYU","target":"record","payload":{"canonical_record":{"source":{"id":"2206.02231","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-06-05T17:58:02Z","cross_cats_sorted":["cs.AI","cs.SY","eess.SY"],"title_canon_sha256":"d82b36090e1b884c71f2701820d7024483dc079f2399abba4172f6175fa0021c","abstract_canon_sha256":"fdb2a99e088a783203be25b87a1227d696195f30168f4dfcf3429c355272874b"},"schema_version":"1.0"},"canonical_sha256":"dd745a3aa65a809a279acf6eb2bcb2c52256d8db190c38cf1a1eb000c3dc4f8d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:48:22.609477Z","signature_b64":"P8HWK1vLzhxJgUlIhBduL3S+gb4fU2jVqCSkl4nqo5/gtU68Ct6PiW6GMZAKqv7uRIp/zliL1pZylpOG5tysBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dd745a3aa65a809a279acf6eb2bcb2c52256d8db190c38cf1a1eb000c3dc4f8d","last_reissued_at":"2026-07-05T06:48:22.608965Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:48:22.608965Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2206.02231","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:48:22Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"lVNSu4HfdoOI23r/f+NqtYaj/P7OS8Y/p7oDwX5BElrCIAn+cSZ17NEpynP45ik5f4ueccpEO7G2hQIvJcVLAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T18:26:41.710092Z"},"content_sha256":"257bf7405d01ff6d4ed804a7deb66e79f7eda2f4e10cf1cf5d8e64a447255a80","schema_version":"1.0","event_id":"sha256:257bf7405d01ff6d4ed804a7deb66e79f7eda2f4e10cf1cf5d8e64a447255a80"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:3V2FUOVGLKAJUJ42Z5XLFPFSYU","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Models of human preference for learning reward functions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Alessandro Allievi, Peter Stone, Scott Niekum, Serena Booth, Stephane Hatgis-Kessell, W. Bradley Knox","submitted_at":"2022-06-05T17:58:02Z","abstract_excerpt":"The utility of reinforcement learning is limited by the alignment of reward functions with the interests of human stakeholders. One promising method for alignment is to learn the reward function from human-generated preferences between pairs of trajectory segments, a type of reinforcement learning from human feedback (RLHF). These human preferences are typically assumed to be informed solely by partial return, the sum of rewards along each segment. We find this assumption to be flawed and propose modeling human preferences instead as informed by each segment's regret, a measure of a segment's "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.02231","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.02231/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:48:22Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"B1bvdSNHQOAFm3RrYnUb/yKOPnirMlgPrH38C42/Rn+bTtwB7EanFUh3jtK1Pn1CeW1P/AHcEmYzTCQO/y/XBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T18:26:41.714611Z"},"content_sha256":"2b8dbde5b35a14b45824f64119d9f266b878de8ee4ebc3ca032f85b4dc822d7c","schema_version":"1.0","event_id":"sha256:2b8dbde5b35a14b45824f64119d9f266b878de8ee4ebc3ca032f85b4dc822d7c"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU/bundle.json","state_url":"https://pith.science/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T18:26:41Z","links":{"resolver":"https://pith.science/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU","bundle":"https://pith.science/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU/bundle.json","state":"https://pith.science/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU/state.json","well_known_bundle":"https://pith.science/.well-known/pith/3V2FUOVGLKAJUJ42Z5XLFPFSYU/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:3V2FUOVGLKAJUJ42Z5XLFPFSYU","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"fdb2a99e088a783203be25b87a1227d696195f30168f4dfcf3429c355272874b","cross_cats_sorted":["cs.AI","cs.SY","eess.SY"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-06-05T17:58:02Z","title_canon_sha256":"d82b36090e1b884c71f2701820d7024483dc079f2399abba4172f6175fa0021c"},"schema_version":"1.0","source":{"id":"2206.02231","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2206.02231","created_at":"2026-07-05T06:48:22Z"},{"alias_kind":"arxiv_version","alias_value":"2206.02231v3","created_at":"2026-07-05T06:48:22Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.02231","created_at":"2026-07-05T06:48:22Z"},{"alias_kind":"pith_short_12","alias_value":"3V2FUOVGLKAJ","created_at":"2026-07-05T06:48:22Z"},{"alias_kind":"pith_short_16","alias_value":"3V2FUOVGLKAJUJ42","created_at":"2026-07-05T06:48:22Z"},{"alias_kind":"pith_short_8","alias_value":"3V2FUOVG","created_at":"2026-07-05T06:48:22Z"}],"graph_snapshots":[{"event_id":"sha256:2b8dbde5b35a14b45824f64119d9f266b878de8ee4ebc3ca032f85b4dc822d7c","target":"graph","created_at":"2026-07-05T06:48:22Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2206.02231/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The utility of reinforcement learning is limited by the alignment of reward functions with the interests of human stakeholders. One promising method for alignment is to learn the reward function from human-generated preferences between pairs of trajectory segments, a type of reinforcement learning from human feedback (RLHF). These human preferences are typically assumed to be informed solely by partial return, the sum of rewards along each segment. We find this assumption to be flawed and propose modeling human preferences instead as informed by each segment's regret, a measure of a segment's ","authors_text":"Alessandro Allievi, Peter Stone, Scott Niekum, Serena Booth, Stephane Hatgis-Kessell, W. Bradley Knox","cross_cats":["cs.AI","cs.SY","eess.SY"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-06-05T17:58:02Z","title":"Models of human preference for learning reward functions"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.02231","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:257bf7405d01ff6d4ed804a7deb66e79f7eda2f4e10cf1cf5d8e64a447255a80","target":"record","created_at":"2026-07-05T06:48:22Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"fdb2a99e088a783203be25b87a1227d696195f30168f4dfcf3429c355272874b","cross_cats_sorted":["cs.AI","cs.SY","eess.SY"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-06-05T17:58:02Z","title_canon_sha256":"d82b36090e1b884c71f2701820d7024483dc079f2399abba4172f6175fa0021c"},"schema_version":"1.0","source":{"id":"2206.02231","kind":"arxiv","version":3}},"canonical_sha256":"dd745a3aa65a809a279acf6eb2bcb2c52256d8db190c38cf1a1eb000c3dc4f8d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"dd745a3aa65a809a279acf6eb2bcb2c52256d8db190c38cf1a1eb000c3dc4f8d","first_computed_at":"2026-07-05T06:48:22.608965Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:48:22.608965Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"P8HWK1vLzhxJgUlIhBduL3S+gb4fU2jVqCSkl4nqo5/gtU68Ct6PiW6GMZAKqv7uRIp/zliL1pZylpOG5tysBA==","signature_status":"signed_v1","signed_at":"2026-07-05T06:48:22.609477Z","signed_message":"canonical_sha256_bytes"},"source_id":"2206.02231","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:257bf7405d01ff6d4ed804a7deb66e79f7eda2f4e10cf1cf5d8e64a447255a80","sha256:2b8dbde5b35a14b45824f64119d9f266b878de8ee4ebc3ca032f85b4dc822d7c"],"state_sha256":"54b613a10926da4a82717e7efd90529286e68c10fe78a1e678f3c69bf83cd220"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"rTzxE5N5ESI8O+Mc/gCr0bPJgn3swH+cF7osdGyVFR8sF/f7mRLiiRv0GQei0o0WWBFphcPgZusuCGl125sVDw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T18:26:41.721496Z","bundle_sha256":"acd25133077c1f84cf68e6052220b61cbeb16ec9b5674132d3f2bd9ba443af16"}}